Files
analise-artigo/analise_artigo/analyze.py
Jário José 550440f4a7 Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
2026-09-25 13:24:56 -03:00

347 lines
12 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Métricas textuais, consistência de citações e checklist de conformidade.
Regras ABNT/NBR usadas:
- NBR 10520: citações no texto devem existir na referências e vice-versa;
- NBR 6022: elementos obrigatórios do artigo (título, resumo,
palavras-chave, seções, referências);
- NBR 14724: elementos pré/textuais/pós-textuais.
"""
from __future__ import annotations
import re
from collections import Counter
from dataclasses import dataclass, field
from .bib import Reference
from .texparse import Article, Section, count_words
# ---------------------------------------------------------------------------
# Estruturas de resultado
# ---------------------------------------------------------------------------
@dataclass
class Check:
code: str
label: str
status: str # ok | warn | fail
detail: str = ""
@property
def icon(self) -> str:
return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status]
@dataclass
class Analysis:
article: Article
refs: dict[str, Reference]
checks: list[Check] = field(default_factory=list)
metrics: dict = field(default_factory=dict)
issues: list[str] = field(default_factory=list)
# ---------------------------------------------------------------------------
# Métricas de prosa
# ---------------------------------------------------------------------------
VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]"
def syllables(word: str) -> int:
"""Aproximação de sílabas: 1 sílaba por grupo de vogais.
Contar *grupos* (em vez de caracteres) evita inflar ditongos como
"ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto).
"""
w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower())
if not w:
return 1
n = len(re.findall(VOWELS + "+", w))
return max(1, n)
def sentences(text: str) -> list[str]:
"""Divide em frases respeitando as quebras de linha do .tex.
Nos artigos processados cada parágrafo/item de lista ocupa uma linha
própria (convenção verificada nas fontes), então linha vira fronteira
segura de frase; dentro da linha, pontuação [.!?] segue valendo.
"""
out: list[str] = []
for line in text.splitlines():
line = line.strip()
if not line:
continue
parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line)
for p in parts:
p = p.strip()
# ignora marcadores de lista ("-", "1.") e ruído trivial
if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p):
continue
out.append(p)
return out
def flesch(text: str) -> float | None:
"""Índice de Flesch adaptado para o português (0-100; mais alto = mais legível).
Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com
sílabas estimadas por grupos de vogais (aproximação, sem acentuação
completa). Resultado clamped em [0, 100].
"""
words = count_words(text)
sents = sentences(text)
if not words or not sents:
return None
syl = sum(syllables(w) for w in words)
mpf = len(words) / len(sents) # média de palavras por frase
vmp = syl / len(words) # vocabulário médio (sílabas/palavra)
score = 206.835 - 1.015 * mpf - 84.6 * vmp
return round(min(100.0, max(0.0, score)), 1)
def flesch_band(score: float | None) -> str:
if score is None:
return "n/d"
if score >= 80:
return "Muito fácil"
if score >= 60:
return "Fácil"
if score >= 40:
return "Médio"
if score >= 20:
return "Difícil"
return "Muito difícil"
# ---------------------------------------------------------------------------
# Heurísticas de qualidade
# ---------------------------------------------------------------------------
REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I)
REDUNDANT = re.compile(
r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I
)
LONG_SENTENCE = 60 # palavras
LONG_PARAGRAPH = 160 # palavras
def find_grammar_issues(text: str) -> list[str]:
issues: list[str] = []
for m in REPEATED_WORD.finditer(text):
# "se se" é legítimo em português pronominal ("avaliou-se se cada
# estudo..."); filtrado para não gerar falso-positivo
if m.group(1).lower() == "se":
continue
issues.append(
f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — "
f"...{context_around(text, m.start())}..."
)
for m in REDUNDANT.finditer(text):
issues.append(
f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — "
f"...{context_around(text, m.start())}..."
)
return issues
def context_around(text: str, pos: int, span: int = 60) -> str:
start = max(0, pos - span)
end = min(len(text), pos + span)
snippet = text[start:end].replace("\n", " ")
return ("…" + snippet) if start > 0 else snippet
def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]:
out = []
for s in sentences(text):
n = len(count_words(s))
if n > limit:
out.append((s[:120] + ("…" if len(s) > 120 else ""), n))
return sorted(out, key=lambda x: -x[1])[:10]
def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]:
# Convenção das fontes analisadas: um parágrafo por linha no .tex,
# portanto cada linha (depois da limpeza) é candidata a parágrafo.
out = []
for p in (line.strip() for line in text.splitlines()):
if not p:
continue
n = len(count_words(p))
if n > limit:
out.append((p[:120] + "…", n))
return sorted(out, key=lambda x: -x[1])[:10]
def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]:
words = [w.lower() for w in count_words(text)]
stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os",
"as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são",
"and", "or"} # operadores de busca do estilo PRISMA (AND/OR)
words = [w for w in words if w not in stop and len(w) > 2]
grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)]
return Counter(grams).most_common(top)
def citation_stats(article: Article, refs: dict[str, Reference]) -> dict:
cited = Counter(article.citations)
defined = set(refs)
used = set(cited)
return {
"total_citations": len(article.citations),
"unique_cited": len(used),
"defined_refs": len(defined),
"orphan_citations": sorted(used - defined),
"unused_refs": sorted(defined - used),
"most_cited": [
(k, n, refs.get(k).label() if k in refs else "?")
for k, n in cited.most_common(8)
],
"year_distribution": dict(
Counter(
(refs[k].year or "?") if k in refs else "?" for k in used
)
),
}
# ---------------------------------------------------------------------------
# Análise completa
# ---------------------------------------------------------------------------
RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"]
EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura",
"estado da arte", "revisão de literatura",
"fundamentação", "referencial"]
def analyze(article: Article, refs: dict[str, Reference]) -> Analysis:
checks: list[Check] = []
issues: list[str] = []
# --- estrutura / pré-textuais ----------------------------------------
checks.append(Check(
"ABNT-01", "Título presente",
"ok" if article.title else "fail",
article.title or "não encontrado no \\titulo{}",
))
checks.append(Check(
"ABNT-02", "Autores informados",
"ok" if article.authors else "fail",
", ".join(article.authors) or "não encontrado",
))
checks.append(Check(
"ABNT-03", "Local e data",
"ok" if article.local and article.data else "warn",
f"{article.local}, {article.data}".strip(",") or "incompleto",
))
abs_words = len(count_words(article.abstract))
checks.append(Check(
"ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)",
"ok" if 150 <= abs_words <= 500 else "warn",
f"{abs_words} palavras" if abs_words else "não encontrado",
))
checks.append(Check(
"ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)",
"ok" if 3 <= len(article.keywords) <= 8 else "warn",
", ".join(article.keywords) or "não encontradas",
))
all_titles = [s.title.lower() for s in article.sections]
for expected in RECOGNIZED_SECTIONS:
hit = next((t for t in all_titles if expected in t), None)
checks.append(Check(
"ABNT-06", f"Seção '{expected.capitalize()}'",
"ok" if hit else "warn",
hit or "não identificada",
))
lit_hit = next((t for t in all_titles
if any(e in t for e in EXPECTED_LITERATURE)), None)
checks.append(Check(
"ABNT-07", "Seção de revisão de literatura",
"ok" if lit_hit else "fail",
lit_hit or "faltando",
))
# citações / referências ------------------------------------------------
stats = citation_stats(article, refs)
checks.append(Check(
"ABNT-08", "Citações no corpo do texto",
"ok" if stats["total_citations"] >= 5 else "warn",
f"{stats['total_citations']} citações, "
f"{stats['unique_cited']} referências citadas",
))
checks.append(Check(
"ABNT-09", "Citações sem entrada no .bib (NBR 10520)",
"ok" if not stats["orphan_citations"] else "fail",
", ".join(stats["orphan_citations"]) or "nenhuma",
))
checks.append(Check(
"ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)",
"ok" if not stats["unused_refs"] else "warn",
", ".join(stats["unused_refs"]) or "nenhuma",
))
# figuras ------------------------------------------------------------
checks.append(Check(
"ABNT-11", "Figuras/quadros identificados",
"ok" if (article.figures or article.tables) else "warn",
f"{len(article.figures)} figuras, {len(article.tables)} quadros",
))
# prosa ---------------------------------------------------------------
body_text = "\n\n".join(s.text for s in article.sections)
flesch_score = flesch(body_text)
top_phrases = frequent_phrases(body_text)
repeated = find_grammar_issues(article.abstract + " " + body_text)
l_sentences = long_sentences(body_text)
l_paragraphs = long_paragraphs(body_text)
for i in repeated:
issues.append(i)
for s, n in l_sentences:
issues.append(f"Frase longa ({n} palavras): \"{s}\"")
for p, n in l_paragraphs:
issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"")
metrics = {
"palavras_corpo": len(count_words(body_text)),
"palavras_totais": article.text_words,
"sentencas": len(sentences(body_text)),
"media_palavras_por_sentenca": round(
len(count_words(body_text)) / max(1, len(sentences(body_text))), 1
),
"vocabulario_unico": len(
set(w.lower() for w in count_words(body_text))
),
"flesch": flesch_score,
"flesch_faixa": flesch_band(flesch_score),
"secoes": [
{
"titulo": s.title,
"nivel": s.level,
"palavras": s.words,
"percentual": round(
100 * s.words / max(1, article.text_words), 1
),
}
for s in article.sections
],
"top_frases": top_phrases,
"citacoes": stats,
}
summary = {
status: sum(1 for c in checks if c.status == status)
for status in ("ok", "warn", "fail")
}
metrics["resumo_checks"] = summary
return Analysis(
article=article, refs=refs, checks=checks, metrics=metrics,
issues=issues,
)