Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
346
analise_artigo/analyze.py
Normal file
346
analise_artigo/analyze.py
Normal file
@@ -0,0 +1,346 @@
|
||||
"""Métricas textuais, consistência de citações e checklist de conformidade.
|
||||
|
||||
Regras ABNT/NBR usadas:
|
||||
- NBR 10520: citações no texto devem existir na referências e vice-versa;
|
||||
- NBR 6022: elementos obrigatórios do artigo (título, resumo,
|
||||
palavras-chave, seções, referências);
|
||||
- NBR 14724: elementos pré/textuais/pós-textuais.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from .bib import Reference
|
||||
from .texparse import Article, Section, count_words
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Estruturas de resultado
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@dataclass
|
||||
class Check:
|
||||
code: str
|
||||
label: str
|
||||
status: str # ok | warn | fail
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def icon(self) -> str:
|
||||
return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status]
|
||||
|
||||
|
||||
@dataclass
|
||||
class Analysis:
|
||||
article: Article
|
||||
refs: dict[str, Reference]
|
||||
checks: list[Check] = field(default_factory=list)
|
||||
metrics: dict = field(default_factory=dict)
|
||||
issues: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Métricas de prosa
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]"
|
||||
|
||||
|
||||
def syllables(word: str) -> int:
|
||||
"""Aproximação de sílabas: 1 sílaba por grupo de vogais.
|
||||
|
||||
Contar *grupos* (em vez de caracteres) evita inflar ditongos como
|
||||
"ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto).
|
||||
"""
|
||||
w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower())
|
||||
if not w:
|
||||
return 1
|
||||
n = len(re.findall(VOWELS + "+", w))
|
||||
return max(1, n)
|
||||
|
||||
|
||||
def sentences(text: str) -> list[str]:
|
||||
"""Divide em frases respeitando as quebras de linha do .tex.
|
||||
|
||||
Nos artigos processados cada parágrafo/item de lista ocupa uma linha
|
||||
própria (convenção verificada nas fontes), então linha vira fronteira
|
||||
segura de frase; dentro da linha, pontuação [.!?] segue valendo.
|
||||
"""
|
||||
out: list[str] = []
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line)
|
||||
for p in parts:
|
||||
p = p.strip()
|
||||
# ignora marcadores de lista ("-", "1.") e ruído trivial
|
||||
if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p):
|
||||
continue
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
def flesch(text: str) -> float | None:
|
||||
"""Índice de Flesch adaptado para o português (0-100; mais alto = mais legível).
|
||||
|
||||
Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com
|
||||
sílabas estimadas por grupos de vogais (aproximação, sem acentuação
|
||||
completa). Resultado clamped em [0, 100].
|
||||
"""
|
||||
words = count_words(text)
|
||||
sents = sentences(text)
|
||||
if not words or not sents:
|
||||
return None
|
||||
syl = sum(syllables(w) for w in words)
|
||||
mpf = len(words) / len(sents) # média de palavras por frase
|
||||
vmp = syl / len(words) # vocabulário médio (sílabas/palavra)
|
||||
score = 206.835 - 1.015 * mpf - 84.6 * vmp
|
||||
return round(min(100.0, max(0.0, score)), 1)
|
||||
|
||||
|
||||
def flesch_band(score: float | None) -> str:
|
||||
if score is None:
|
||||
return "n/d"
|
||||
if score >= 80:
|
||||
return "Muito fácil"
|
||||
if score >= 60:
|
||||
return "Fácil"
|
||||
if score >= 40:
|
||||
return "Médio"
|
||||
if score >= 20:
|
||||
return "Difícil"
|
||||
return "Muito difícil"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Heurísticas de qualidade
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I)
|
||||
REDUNDANT = re.compile(
|
||||
r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I
|
||||
)
|
||||
LONG_SENTENCE = 60 # palavras
|
||||
LONG_PARAGRAPH = 160 # palavras
|
||||
|
||||
|
||||
def find_grammar_issues(text: str) -> list[str]:
|
||||
issues: list[str] = []
|
||||
for m in REPEATED_WORD.finditer(text):
|
||||
# "se se" é legítimo em português pronominal ("avaliou-se se cada
|
||||
# estudo..."); filtrado para não gerar falso-positivo
|
||||
if m.group(1).lower() == "se":
|
||||
continue
|
||||
issues.append(
|
||||
f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — "
|
||||
f"...{context_around(text, m.start())}..."
|
||||
)
|
||||
for m in REDUNDANT.finditer(text):
|
||||
issues.append(
|
||||
f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — "
|
||||
f"...{context_around(text, m.start())}..."
|
||||
)
|
||||
return issues
|
||||
|
||||
|
||||
def context_around(text: str, pos: int, span: int = 60) -> str:
|
||||
start = max(0, pos - span)
|
||||
end = min(len(text), pos + span)
|
||||
snippet = text[start:end].replace("\n", " ")
|
||||
return ("…" + snippet) if start > 0 else snippet
|
||||
|
||||
|
||||
def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]:
|
||||
out = []
|
||||
for s in sentences(text):
|
||||
n = len(count_words(s))
|
||||
if n > limit:
|
||||
out.append((s[:120] + ("…" if len(s) > 120 else ""), n))
|
||||
return sorted(out, key=lambda x: -x[1])[:10]
|
||||
|
||||
|
||||
def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]:
|
||||
# Convenção das fontes analisadas: um parágrafo por linha no .tex,
|
||||
# portanto cada linha (depois da limpeza) é candidata a parágrafo.
|
||||
out = []
|
||||
for p in (line.strip() for line in text.splitlines()):
|
||||
if not p:
|
||||
continue
|
||||
n = len(count_words(p))
|
||||
if n > limit:
|
||||
out.append((p[:120] + "…", n))
|
||||
return sorted(out, key=lambda x: -x[1])[:10]
|
||||
|
||||
|
||||
def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]:
|
||||
words = [w.lower() for w in count_words(text)]
|
||||
stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os",
|
||||
"as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são",
|
||||
"and", "or"} # operadores de busca do estilo PRISMA (AND/OR)
|
||||
words = [w for w in words if w not in stop and len(w) > 2]
|
||||
grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)]
|
||||
return Counter(grams).most_common(top)
|
||||
|
||||
|
||||
def citation_stats(article: Article, refs: dict[str, Reference]) -> dict:
|
||||
cited = Counter(article.citations)
|
||||
defined = set(refs)
|
||||
used = set(cited)
|
||||
return {
|
||||
"total_citations": len(article.citations),
|
||||
"unique_cited": len(used),
|
||||
"defined_refs": len(defined),
|
||||
"orphan_citations": sorted(used - defined),
|
||||
"unused_refs": sorted(defined - used),
|
||||
"most_cited": [
|
||||
(k, n, refs.get(k).label() if k in refs else "?")
|
||||
for k, n in cited.most_common(8)
|
||||
],
|
||||
"year_distribution": dict(
|
||||
Counter(
|
||||
(refs[k].year or "?") if k in refs else "?" for k in used
|
||||
)
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Análise completa
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"]
|
||||
EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura",
|
||||
"estado da arte", "revisão de literatura",
|
||||
"fundamentação", "referencial"]
|
||||
|
||||
|
||||
def analyze(article: Article, refs: dict[str, Reference]) -> Analysis:
|
||||
checks: list[Check] = []
|
||||
issues: list[str] = []
|
||||
|
||||
# --- estrutura / pré-textuais ----------------------------------------
|
||||
checks.append(Check(
|
||||
"ABNT-01", "Título presente",
|
||||
"ok" if article.title else "fail",
|
||||
article.title or "não encontrado no \\titulo{}",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-02", "Autores informados",
|
||||
"ok" if article.authors else "fail",
|
||||
", ".join(article.authors) or "não encontrado",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-03", "Local e data",
|
||||
"ok" if article.local and article.data else "warn",
|
||||
f"{article.local}, {article.data}".strip(",") or "incompleto",
|
||||
))
|
||||
|
||||
abs_words = len(count_words(article.abstract))
|
||||
checks.append(Check(
|
||||
"ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)",
|
||||
"ok" if 150 <= abs_words <= 500 else "warn",
|
||||
f"{abs_words} palavras" if abs_words else "não encontrado",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)",
|
||||
"ok" if 3 <= len(article.keywords) <= 8 else "warn",
|
||||
", ".join(article.keywords) or "não encontradas",
|
||||
))
|
||||
|
||||
all_titles = [s.title.lower() for s in article.sections]
|
||||
for expected in RECOGNIZED_SECTIONS:
|
||||
hit = next((t for t in all_titles if expected in t), None)
|
||||
checks.append(Check(
|
||||
"ABNT-06", f"Seção '{expected.capitalize()}'",
|
||||
"ok" if hit else "warn",
|
||||
hit or "não identificada",
|
||||
))
|
||||
lit_hit = next((t for t in all_titles
|
||||
if any(e in t for e in EXPECTED_LITERATURE)), None)
|
||||
checks.append(Check(
|
||||
"ABNT-07", "Seção de revisão de literatura",
|
||||
"ok" if lit_hit else "fail",
|
||||
lit_hit or "faltando",
|
||||
))
|
||||
|
||||
# citações / referências ------------------------------------------------
|
||||
stats = citation_stats(article, refs)
|
||||
checks.append(Check(
|
||||
"ABNT-08", "Citações no corpo do texto",
|
||||
"ok" if stats["total_citations"] >= 5 else "warn",
|
||||
f"{stats['total_citations']} citações, "
|
||||
f"{stats['unique_cited']} referências citadas",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-09", "Citações sem entrada no .bib (NBR 10520)",
|
||||
"ok" if not stats["orphan_citations"] else "fail",
|
||||
", ".join(stats["orphan_citations"]) or "nenhuma",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)",
|
||||
"ok" if not stats["unused_refs"] else "warn",
|
||||
", ".join(stats["unused_refs"]) or "nenhuma",
|
||||
))
|
||||
|
||||
# figuras ------------------------------------------------------------
|
||||
checks.append(Check(
|
||||
"ABNT-11", "Figuras/quadros identificados",
|
||||
"ok" if (article.figures or article.tables) else "warn",
|
||||
f"{len(article.figures)} figuras, {len(article.tables)} quadros",
|
||||
))
|
||||
|
||||
# prosa ---------------------------------------------------------------
|
||||
body_text = "\n\n".join(s.text for s in article.sections)
|
||||
flesch_score = flesch(body_text)
|
||||
top_phrases = frequent_phrases(body_text)
|
||||
repeated = find_grammar_issues(article.abstract + " " + body_text)
|
||||
l_sentences = long_sentences(body_text)
|
||||
l_paragraphs = long_paragraphs(body_text)
|
||||
for i in repeated:
|
||||
issues.append(i)
|
||||
for s, n in l_sentences:
|
||||
issues.append(f"Frase longa ({n} palavras): \"{s}\"")
|
||||
for p, n in l_paragraphs:
|
||||
issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"")
|
||||
|
||||
metrics = {
|
||||
"palavras_corpo": len(count_words(body_text)),
|
||||
"palavras_totais": article.text_words,
|
||||
"sentencas": len(sentences(body_text)),
|
||||
"media_palavras_por_sentenca": round(
|
||||
len(count_words(body_text)) / max(1, len(sentences(body_text))), 1
|
||||
),
|
||||
"vocabulario_unico": len(
|
||||
set(w.lower() for w in count_words(body_text))
|
||||
),
|
||||
"flesch": flesch_score,
|
||||
"flesch_faixa": flesch_band(flesch_score),
|
||||
"secoes": [
|
||||
{
|
||||
"titulo": s.title,
|
||||
"nivel": s.level,
|
||||
"palavras": s.words,
|
||||
"percentual": round(
|
||||
100 * s.words / max(1, article.text_words), 1
|
||||
),
|
||||
}
|
||||
for s in article.sections
|
||||
],
|
||||
"top_frases": top_phrases,
|
||||
"citacoes": stats,
|
||||
}
|
||||
|
||||
summary = {
|
||||
status: sum(1 for c in checks if c.status == status)
|
||||
for status in ("ok", "warn", "fail")
|
||||
}
|
||||
metrics["resumo_checks"] = summary
|
||||
|
||||
return Analysis(
|
||||
article=article, refs=refs, checks=checks, metrics=metrics,
|
||||
issues=issues,
|
||||
)
|
||||
Reference in New Issue
Block a user