Projeto analise-artigo: análise ABNT de artigos LaTeX

- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
2026-09-25 13:24:56 -03:00
commit 550440f4a7
14 changed files with 1591 additions and 0 deletions

346
analise_artigo/analyze.py Normal file
View File

@@ -0,0 +1,346 @@
"""Métricas textuais, consistência de citações e checklist de conformidade.
Regras ABNT/NBR usadas:
- NBR 10520: citações no texto devem existir na referências e vice-versa;
- NBR 6022: elementos obrigatórios do artigo (título, resumo,
palavras-chave, seções, referências);
- NBR 14724: elementos pré/textuais/pós-textuais.
"""
from __future__ import annotations
import re
from collections import Counter
from dataclasses import dataclass, field
from .bib import Reference
from .texparse import Article, Section, count_words
# ---------------------------------------------------------------------------
# Estruturas de resultado
# ---------------------------------------------------------------------------
@dataclass
class Check:
code: str
label: str
status: str # ok | warn | fail
detail: str = ""
@property
def icon(self) -> str:
return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status]
@dataclass
class Analysis:
article: Article
refs: dict[str, Reference]
checks: list[Check] = field(default_factory=list)
metrics: dict = field(default_factory=dict)
issues: list[str] = field(default_factory=list)
# ---------------------------------------------------------------------------
# Métricas de prosa
# ---------------------------------------------------------------------------
VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]"
def syllables(word: str) -> int:
"""Aproximação de sílabas: 1 sílaba por grupo de vogais.
Contar *grupos* (em vez de caracteres) evita inflar ditongos como
"ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto).
"""
w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower())
if not w:
return 1
n = len(re.findall(VOWELS + "+", w))
return max(1, n)
def sentences(text: str) -> list[str]:
"""Divide em frases respeitando as quebras de linha do .tex.
Nos artigos processados cada parágrafo/item de lista ocupa uma linha
própria (convenção verificada nas fontes), então linha vira fronteira
segura de frase; dentro da linha, pontuação [.!?] segue valendo.
"""
out: list[str] = []
for line in text.splitlines():
line = line.strip()
if not line:
continue
parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line)
for p in parts:
p = p.strip()
# ignora marcadores de lista ("-", "1.") e ruído trivial
if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p):
continue
out.append(p)
return out
def flesch(text: str) -> float | None:
"""Índice de Flesch adaptado para o português (0-100; mais alto = mais legível).
Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com
sílabas estimadas por grupos de vogais (aproximação, sem acentuação
completa). Resultado clamped em [0, 100].
"""
words = count_words(text)
sents = sentences(text)
if not words or not sents:
return None
syl = sum(syllables(w) for w in words)
mpf = len(words) / len(sents) # média de palavras por frase
vmp = syl / len(words) # vocabulário médio (sílabas/palavra)
score = 206.835 - 1.015 * mpf - 84.6 * vmp
return round(min(100.0, max(0.0, score)), 1)
def flesch_band(score: float | None) -> str:
if score is None:
return "n/d"
if score >= 80:
return "Muito fácil"
if score >= 60:
return "Fácil"
if score >= 40:
return "Médio"
if score >= 20:
return "Difícil"
return "Muito difícil"
# ---------------------------------------------------------------------------
# Heurísticas de qualidade
# ---------------------------------------------------------------------------
REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I)
REDUNDANT = re.compile(
r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I
)
LONG_SENTENCE = 60 # palavras
LONG_PARAGRAPH = 160 # palavras
def find_grammar_issues(text: str) -> list[str]:
issues: list[str] = []
for m in REPEATED_WORD.finditer(text):
# "se se" é legítimo em português pronominal ("avaliou-se se cada
# estudo..."); filtrado para não gerar falso-positivo
if m.group(1).lower() == "se":
continue
issues.append(
f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — "
f"...{context_around(text, m.start())}..."
)
for m in REDUNDANT.finditer(text):
issues.append(
f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — "
f"...{context_around(text, m.start())}..."
)
return issues
def context_around(text: str, pos: int, span: int = 60) -> str:
start = max(0, pos - span)
end = min(len(text), pos + span)
snippet = text[start:end].replace("\n", " ")
return ("…" + snippet) if start > 0 else snippet
def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]:
out = []
for s in sentences(text):
n = len(count_words(s))
if n > limit:
out.append((s[:120] + ("…" if len(s) > 120 else ""), n))
return sorted(out, key=lambda x: -x[1])[:10]
def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]:
# Convenção das fontes analisadas: um parágrafo por linha no .tex,
# portanto cada linha (depois da limpeza) é candidata a parágrafo.
out = []
for p in (line.strip() for line in text.splitlines()):
if not p:
continue
n = len(count_words(p))
if n > limit:
out.append((p[:120] + "…", n))
return sorted(out, key=lambda x: -x[1])[:10]
def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]:
words = [w.lower() for w in count_words(text)]
stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os",
"as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são",
"and", "or"} # operadores de busca do estilo PRISMA (AND/OR)
words = [w for w in words if w not in stop and len(w) > 2]
grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)]
return Counter(grams).most_common(top)
def citation_stats(article: Article, refs: dict[str, Reference]) -> dict:
cited = Counter(article.citations)
defined = set(refs)
used = set(cited)
return {
"total_citations": len(article.citations),
"unique_cited": len(used),
"defined_refs": len(defined),
"orphan_citations": sorted(used - defined),
"unused_refs": sorted(defined - used),
"most_cited": [
(k, n, refs.get(k).label() if k in refs else "?")
for k, n in cited.most_common(8)
],
"year_distribution": dict(
Counter(
(refs[k].year or "?") if k in refs else "?" for k in used
)
),
}
# ---------------------------------------------------------------------------
# Análise completa
# ---------------------------------------------------------------------------
RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"]
EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura",
"estado da arte", "revisão de literatura",
"fundamentação", "referencial"]
def analyze(article: Article, refs: dict[str, Reference]) -> Analysis:
checks: list[Check] = []
issues: list[str] = []
# --- estrutura / pré-textuais ----------------------------------------
checks.append(Check(
"ABNT-01", "Título presente",
"ok" if article.title else "fail",
article.title or "não encontrado no \\titulo{}",
))
checks.append(Check(
"ABNT-02", "Autores informados",
"ok" if article.authors else "fail",
", ".join(article.authors) or "não encontrado",
))
checks.append(Check(
"ABNT-03", "Local e data",
"ok" if article.local and article.data else "warn",
f"{article.local}, {article.data}".strip(",") or "incompleto",
))
abs_words = len(count_words(article.abstract))
checks.append(Check(
"ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)",
"ok" if 150 <= abs_words <= 500 else "warn",
f"{abs_words} palavras" if abs_words else "não encontrado",
))
checks.append(Check(
"ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)",
"ok" if 3 <= len(article.keywords) <= 8 else "warn",
", ".join(article.keywords) or "não encontradas",
))
all_titles = [s.title.lower() for s in article.sections]
for expected in RECOGNIZED_SECTIONS:
hit = next((t for t in all_titles if expected in t), None)
checks.append(Check(
"ABNT-06", f"Seção '{expected.capitalize()}'",
"ok" if hit else "warn",
hit or "não identificada",
))
lit_hit = next((t for t in all_titles
if any(e in t for e in EXPECTED_LITERATURE)), None)
checks.append(Check(
"ABNT-07", "Seção de revisão de literatura",
"ok" if lit_hit else "fail",
lit_hit or "faltando",
))
# citações / referências ------------------------------------------------
stats = citation_stats(article, refs)
checks.append(Check(
"ABNT-08", "Citações no corpo do texto",
"ok" if stats["total_citations"] >= 5 else "warn",
f"{stats['total_citations']} citações, "
f"{stats['unique_cited']} referências citadas",
))
checks.append(Check(
"ABNT-09", "Citações sem entrada no .bib (NBR 10520)",
"ok" if not stats["orphan_citations"] else "fail",
", ".join(stats["orphan_citations"]) or "nenhuma",
))
checks.append(Check(
"ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)",
"ok" if not stats["unused_refs"] else "warn",
", ".join(stats["unused_refs"]) or "nenhuma",
))
# figuras ------------------------------------------------------------
checks.append(Check(
"ABNT-11", "Figuras/quadros identificados",
"ok" if (article.figures or article.tables) else "warn",
f"{len(article.figures)} figuras, {len(article.tables)} quadros",
))
# prosa ---------------------------------------------------------------
body_text = "\n\n".join(s.text for s in article.sections)
flesch_score = flesch(body_text)
top_phrases = frequent_phrases(body_text)
repeated = find_grammar_issues(article.abstract + " " + body_text)
l_sentences = long_sentences(body_text)
l_paragraphs = long_paragraphs(body_text)
for i in repeated:
issues.append(i)
for s, n in l_sentences:
issues.append(f"Frase longa ({n} palavras): \"{s}\"")
for p, n in l_paragraphs:
issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"")
metrics = {
"palavras_corpo": len(count_words(body_text)),
"palavras_totais": article.text_words,
"sentencas": len(sentences(body_text)),
"media_palavras_por_sentenca": round(
len(count_words(body_text)) / max(1, len(sentences(body_text))), 1
),
"vocabulario_unico": len(
set(w.lower() for w in count_words(body_text))
),
"flesch": flesch_score,
"flesch_faixa": flesch_band(flesch_score),
"secoes": [
{
"titulo": s.title,
"nivel": s.level,
"palavras": s.words,
"percentual": round(
100 * s.words / max(1, article.text_words), 1
),
}
for s in article.sections
],
"top_frases": top_phrases,
"citacoes": stats,
}
summary = {
status: sum(1 for c in checks if c.status == status)
for status in ("ok", "warn", "fail")
}
metrics["resumo_checks"] = summary
return Analysis(
article=article, refs=refs, checks=checks, metrics=metrics,
issues=issues,
)