Projeto analise-artigo: análise ABNT de artigos LaTeX
- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
23
analise_artigo/__init__.py
Normal file
23
analise_artigo/__init__.py
Normal file
@@ -0,0 +1,23 @@
|
||||
"""Análise de artigos acadêmicos em LaTeX (ABNT/abntex2).
|
||||
|
||||
Empacota os módulos de parsing (LaTeX e .bib), métricas,
|
||||
verificações e geração de relatório.
|
||||
"""
|
||||
|
||||
from .texparse import Article, parse_article
|
||||
from .bib import Reference, parse_bib
|
||||
from .analyze import Analysis, analyze
|
||||
from .report import build_markdown, build_json
|
||||
|
||||
__all__ = [
|
||||
"Article",
|
||||
"parse_article",
|
||||
"Reference",
|
||||
"parse_bib",
|
||||
"Analysis",
|
||||
"analyze",
|
||||
"build_markdown",
|
||||
"build_json",
|
||||
]
|
||||
|
||||
__version__ = "1.0.0"
|
||||
85
analise_artigo/__main__.py
Normal file
85
analise_artigo/__main__.py
Normal file
@@ -0,0 +1,85 @@
|
||||
"""Linha de comando.
|
||||
|
||||
Exemplos:
|
||||
python -m analise_artigo artigos/rumo-a-eficiencia --out relatorios
|
||||
python -m analise_artigo caminho/para/main.tex
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
from .bib import parse_bib
|
||||
from .report import write_reports
|
||||
from .texparse import parse_article
|
||||
from .analyze import analyze
|
||||
|
||||
|
||||
def find_tex(target: Path) -> Path | None:
|
||||
if target.is_file() and target.suffix == ".tex":
|
||||
return target
|
||||
if target.is_dir():
|
||||
candidates = sorted(target.glob("*.tex"))
|
||||
return candidates[0] if candidates else None
|
||||
return None
|
||||
|
||||
|
||||
def find_bib(tex: Path) -> Path | None:
|
||||
bib = tex.with_name(tex.stem + ".bib")
|
||||
if bib.exists():
|
||||
return bib
|
||||
for candidate in sorted(tex.parent.glob("*.bib")):
|
||||
return candidate
|
||||
return None
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
p = argparse.ArgumentParser(
|
||||
prog="analise-artigo",
|
||||
description="Analisa um artigo LaTeX (ABNT) e gera relatório de "
|
||||
"estrutura, métricas e citações.",
|
||||
)
|
||||
p.add_argument(
|
||||
"artigo",
|
||||
help="arquivo .tex ou diretório que contém o .tex",
|
||||
)
|
||||
p.add_argument(
|
||||
"--out", "-o",
|
||||
default="relatorios",
|
||||
help="diretório de saída do relatório (padrão: relatorios/)",
|
||||
)
|
||||
args = p.parse_args(argv)
|
||||
|
||||
target = Path(args.artigo)
|
||||
tex = find_tex(target)
|
||||
if tex is None:
|
||||
print(f"erro: nenhum .tex encontrado em {target}", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
bib = find_bib(tex)
|
||||
|
||||
print(f"→ artigo: {tex}")
|
||||
print(f"→ referências: {bib or 'nenhum .bib encontrado'}")
|
||||
|
||||
article = parse_article(tex)
|
||||
refs = parse_bib(bib) if bib else {}
|
||||
|
||||
analysis = analyze(article, refs)
|
||||
|
||||
outdir = Path(args.out)
|
||||
md, js = write_reports(analysis, outdir)
|
||||
|
||||
resumo = analysis.metrics["resumo_checks"]
|
||||
print(f"\nResumo: {resumo['ok']} ok, "
|
||||
f"{resumo['warn']} aviso(s), "
|
||||
f"{resumo['fail']} falha(s), "
|
||||
f"{len(analysis.issues)} questão(ões) de prosa")
|
||||
print(f"\n→ relatório: {md}")
|
||||
print(f"→ dados: {js}")
|
||||
return 0 if resumo["fail"] == 0 else 2
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
346
analise_artigo/analyze.py
Normal file
346
analise_artigo/analyze.py
Normal file
@@ -0,0 +1,346 @@
|
||||
"""Métricas textuais, consistência de citações e checklist de conformidade.
|
||||
|
||||
Regras ABNT/NBR usadas:
|
||||
- NBR 10520: citações no texto devem existir na referências e vice-versa;
|
||||
- NBR 6022: elementos obrigatórios do artigo (título, resumo,
|
||||
palavras-chave, seções, referências);
|
||||
- NBR 14724: elementos pré/textuais/pós-textuais.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from collections import Counter
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from .bib import Reference
|
||||
from .texparse import Article, Section, count_words
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Estruturas de resultado
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
@dataclass
|
||||
class Check:
|
||||
code: str
|
||||
label: str
|
||||
status: str # ok | warn | fail
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def icon(self) -> str:
|
||||
return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status]
|
||||
|
||||
|
||||
@dataclass
|
||||
class Analysis:
|
||||
article: Article
|
||||
refs: dict[str, Reference]
|
||||
checks: list[Check] = field(default_factory=list)
|
||||
metrics: dict = field(default_factory=dict)
|
||||
issues: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Métricas de prosa
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]"
|
||||
|
||||
|
||||
def syllables(word: str) -> int:
|
||||
"""Aproximação de sílabas: 1 sílaba por grupo de vogais.
|
||||
|
||||
Contar *grupos* (em vez de caracteres) evita inflar ditongos como
|
||||
"ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto).
|
||||
"""
|
||||
w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower())
|
||||
if not w:
|
||||
return 1
|
||||
n = len(re.findall(VOWELS + "+", w))
|
||||
return max(1, n)
|
||||
|
||||
|
||||
def sentences(text: str) -> list[str]:
|
||||
"""Divide em frases respeitando as quebras de linha do .tex.
|
||||
|
||||
Nos artigos processados cada parágrafo/item de lista ocupa uma linha
|
||||
própria (convenção verificada nas fontes), então linha vira fronteira
|
||||
segura de frase; dentro da linha, pontuação [.!?] segue valendo.
|
||||
"""
|
||||
out: list[str] = []
|
||||
for line in text.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line)
|
||||
for p in parts:
|
||||
p = p.strip()
|
||||
# ignora marcadores de lista ("-", "1.") e ruído trivial
|
||||
if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p):
|
||||
continue
|
||||
out.append(p)
|
||||
return out
|
||||
|
||||
|
||||
def flesch(text: str) -> float | None:
|
||||
"""Índice de Flesch adaptado para o português (0-100; mais alto = mais legível).
|
||||
|
||||
Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com
|
||||
sílabas estimadas por grupos de vogais (aproximação, sem acentuação
|
||||
completa). Resultado clamped em [0, 100].
|
||||
"""
|
||||
words = count_words(text)
|
||||
sents = sentences(text)
|
||||
if not words or not sents:
|
||||
return None
|
||||
syl = sum(syllables(w) for w in words)
|
||||
mpf = len(words) / len(sents) # média de palavras por frase
|
||||
vmp = syl / len(words) # vocabulário médio (sílabas/palavra)
|
||||
score = 206.835 - 1.015 * mpf - 84.6 * vmp
|
||||
return round(min(100.0, max(0.0, score)), 1)
|
||||
|
||||
|
||||
def flesch_band(score: float | None) -> str:
|
||||
if score is None:
|
||||
return "n/d"
|
||||
if score >= 80:
|
||||
return "Muito fácil"
|
||||
if score >= 60:
|
||||
return "Fácil"
|
||||
if score >= 40:
|
||||
return "Médio"
|
||||
if score >= 20:
|
||||
return "Difícil"
|
||||
return "Muito difícil"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Heurísticas de qualidade
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I)
|
||||
REDUNDANT = re.compile(
|
||||
r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I
|
||||
)
|
||||
LONG_SENTENCE = 60 # palavras
|
||||
LONG_PARAGRAPH = 160 # palavras
|
||||
|
||||
|
||||
def find_grammar_issues(text: str) -> list[str]:
|
||||
issues: list[str] = []
|
||||
for m in REPEATED_WORD.finditer(text):
|
||||
# "se se" é legítimo em português pronominal ("avaliou-se se cada
|
||||
# estudo..."); filtrado para não gerar falso-positivo
|
||||
if m.group(1).lower() == "se":
|
||||
continue
|
||||
issues.append(
|
||||
f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — "
|
||||
f"...{context_around(text, m.start())}..."
|
||||
)
|
||||
for m in REDUNDANT.finditer(text):
|
||||
issues.append(
|
||||
f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — "
|
||||
f"...{context_around(text, m.start())}..."
|
||||
)
|
||||
return issues
|
||||
|
||||
|
||||
def context_around(text: str, pos: int, span: int = 60) -> str:
|
||||
start = max(0, pos - span)
|
||||
end = min(len(text), pos + span)
|
||||
snippet = text[start:end].replace("\n", " ")
|
||||
return ("…" + snippet) if start > 0 else snippet
|
||||
|
||||
|
||||
def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]:
|
||||
out = []
|
||||
for s in sentences(text):
|
||||
n = len(count_words(s))
|
||||
if n > limit:
|
||||
out.append((s[:120] + ("…" if len(s) > 120 else ""), n))
|
||||
return sorted(out, key=lambda x: -x[1])[:10]
|
||||
|
||||
|
||||
def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]:
|
||||
# Convenção das fontes analisadas: um parágrafo por linha no .tex,
|
||||
# portanto cada linha (depois da limpeza) é candidata a parágrafo.
|
||||
out = []
|
||||
for p in (line.strip() for line in text.splitlines()):
|
||||
if not p:
|
||||
continue
|
||||
n = len(count_words(p))
|
||||
if n > limit:
|
||||
out.append((p[:120] + "…", n))
|
||||
return sorted(out, key=lambda x: -x[1])[:10]
|
||||
|
||||
|
||||
def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]:
|
||||
words = [w.lower() for w in count_words(text)]
|
||||
stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os",
|
||||
"as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são",
|
||||
"and", "or"} # operadores de busca do estilo PRISMA (AND/OR)
|
||||
words = [w for w in words if w not in stop and len(w) > 2]
|
||||
grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)]
|
||||
return Counter(grams).most_common(top)
|
||||
|
||||
|
||||
def citation_stats(article: Article, refs: dict[str, Reference]) -> dict:
|
||||
cited = Counter(article.citations)
|
||||
defined = set(refs)
|
||||
used = set(cited)
|
||||
return {
|
||||
"total_citations": len(article.citations),
|
||||
"unique_cited": len(used),
|
||||
"defined_refs": len(defined),
|
||||
"orphan_citations": sorted(used - defined),
|
||||
"unused_refs": sorted(defined - used),
|
||||
"most_cited": [
|
||||
(k, n, refs.get(k).label() if k in refs else "?")
|
||||
for k, n in cited.most_common(8)
|
||||
],
|
||||
"year_distribution": dict(
|
||||
Counter(
|
||||
(refs[k].year or "?") if k in refs else "?" for k in used
|
||||
)
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Análise completa
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"]
|
||||
EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura",
|
||||
"estado da arte", "revisão de literatura",
|
||||
"fundamentação", "referencial"]
|
||||
|
||||
|
||||
def analyze(article: Article, refs: dict[str, Reference]) -> Analysis:
|
||||
checks: list[Check] = []
|
||||
issues: list[str] = []
|
||||
|
||||
# --- estrutura / pré-textuais ----------------------------------------
|
||||
checks.append(Check(
|
||||
"ABNT-01", "Título presente",
|
||||
"ok" if article.title else "fail",
|
||||
article.title or "não encontrado no \\titulo{}",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-02", "Autores informados",
|
||||
"ok" if article.authors else "fail",
|
||||
", ".join(article.authors) or "não encontrado",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-03", "Local e data",
|
||||
"ok" if article.local and article.data else "warn",
|
||||
f"{article.local}, {article.data}".strip(",") or "incompleto",
|
||||
))
|
||||
|
||||
abs_words = len(count_words(article.abstract))
|
||||
checks.append(Check(
|
||||
"ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)",
|
||||
"ok" if 150 <= abs_words <= 500 else "warn",
|
||||
f"{abs_words} palavras" if abs_words else "não encontrado",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)",
|
||||
"ok" if 3 <= len(article.keywords) <= 8 else "warn",
|
||||
", ".join(article.keywords) or "não encontradas",
|
||||
))
|
||||
|
||||
all_titles = [s.title.lower() for s in article.sections]
|
||||
for expected in RECOGNIZED_SECTIONS:
|
||||
hit = next((t for t in all_titles if expected in t), None)
|
||||
checks.append(Check(
|
||||
"ABNT-06", f"Seção '{expected.capitalize()}'",
|
||||
"ok" if hit else "warn",
|
||||
hit or "não identificada",
|
||||
))
|
||||
lit_hit = next((t for t in all_titles
|
||||
if any(e in t for e in EXPECTED_LITERATURE)), None)
|
||||
checks.append(Check(
|
||||
"ABNT-07", "Seção de revisão de literatura",
|
||||
"ok" if lit_hit else "fail",
|
||||
lit_hit or "faltando",
|
||||
))
|
||||
|
||||
# citações / referências ------------------------------------------------
|
||||
stats = citation_stats(article, refs)
|
||||
checks.append(Check(
|
||||
"ABNT-08", "Citações no corpo do texto",
|
||||
"ok" if stats["total_citations"] >= 5 else "warn",
|
||||
f"{stats['total_citations']} citações, "
|
||||
f"{stats['unique_cited']} referências citadas",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-09", "Citações sem entrada no .bib (NBR 10520)",
|
||||
"ok" if not stats["orphan_citations"] else "fail",
|
||||
", ".join(stats["orphan_citations"]) or "nenhuma",
|
||||
))
|
||||
checks.append(Check(
|
||||
"ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)",
|
||||
"ok" if not stats["unused_refs"] else "warn",
|
||||
", ".join(stats["unused_refs"]) or "nenhuma",
|
||||
))
|
||||
|
||||
# figuras ------------------------------------------------------------
|
||||
checks.append(Check(
|
||||
"ABNT-11", "Figuras/quadros identificados",
|
||||
"ok" if (article.figures or article.tables) else "warn",
|
||||
f"{len(article.figures)} figuras, {len(article.tables)} quadros",
|
||||
))
|
||||
|
||||
# prosa ---------------------------------------------------------------
|
||||
body_text = "\n\n".join(s.text for s in article.sections)
|
||||
flesch_score = flesch(body_text)
|
||||
top_phrases = frequent_phrases(body_text)
|
||||
repeated = find_grammar_issues(article.abstract + " " + body_text)
|
||||
l_sentences = long_sentences(body_text)
|
||||
l_paragraphs = long_paragraphs(body_text)
|
||||
for i in repeated:
|
||||
issues.append(i)
|
||||
for s, n in l_sentences:
|
||||
issues.append(f"Frase longa ({n} palavras): \"{s}\"")
|
||||
for p, n in l_paragraphs:
|
||||
issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"")
|
||||
|
||||
metrics = {
|
||||
"palavras_corpo": len(count_words(body_text)),
|
||||
"palavras_totais": article.text_words,
|
||||
"sentencas": len(sentences(body_text)),
|
||||
"media_palavras_por_sentenca": round(
|
||||
len(count_words(body_text)) / max(1, len(sentences(body_text))), 1
|
||||
),
|
||||
"vocabulario_unico": len(
|
||||
set(w.lower() for w in count_words(body_text))
|
||||
),
|
||||
"flesch": flesch_score,
|
||||
"flesch_faixa": flesch_band(flesch_score),
|
||||
"secoes": [
|
||||
{
|
||||
"titulo": s.title,
|
||||
"nivel": s.level,
|
||||
"palavras": s.words,
|
||||
"percentual": round(
|
||||
100 * s.words / max(1, article.text_words), 1
|
||||
),
|
||||
}
|
||||
for s in article.sections
|
||||
],
|
||||
"top_frases": top_phrases,
|
||||
"citacoes": stats,
|
||||
}
|
||||
|
||||
summary = {
|
||||
status: sum(1 for c in checks if c.status == status)
|
||||
for status in ("ok", "warn", "fail")
|
||||
}
|
||||
metrics["resumo_checks"] = summary
|
||||
|
||||
return Analysis(
|
||||
article=article, refs=refs, checks=checks, metrics=metrics,
|
||||
issues=issues,
|
||||
)
|
||||
167
analise_artigo/bib.py
Normal file
167
analise_artigo/bib.py
Normal file
@@ -0,0 +1,167 @@
|
||||
"""Parser simples de arquivos .bib (BibTeX).
|
||||
|
||||
Usa apenas a biblioteca padrão; suporta campos com valores em chaves
|
||||
aninhadas e valores multi-linha.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@dataclass
|
||||
class Reference:
|
||||
key: str
|
||||
type: str # article, inproceedings, misc, ...
|
||||
fields: dict[str, str] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def authors(self) -> str:
|
||||
value = self.fields.get("author", "").strip()
|
||||
# "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author)
|
||||
value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I)
|
||||
return re.sub(r"\s+and\s+", ", ", value, flags=re.I)
|
||||
|
||||
@property
|
||||
def title(self) -> str:
|
||||
return self.fields.get("title", "").strip()
|
||||
|
||||
@property
|
||||
def year(self) -> str:
|
||||
return self.fields.get("year", "").strip()
|
||||
|
||||
|
||||
@property
|
||||
def journal(self) -> str:
|
||||
return self.fields.get(
|
||||
"journal", self.fields.get("booktitle", "")
|
||||
).strip()
|
||||
|
||||
def label(self) -> str:
|
||||
"""Rótulo curto autor (ano)."""
|
||||
first = re.split(r",|and", self.authors)[0].strip()
|
||||
if "," in first:
|
||||
surname = first.split(",")[0].strip()
|
||||
else:
|
||||
surname = last_token(first)
|
||||
return f"{surname} ({self.year})" if self.year else surname
|
||||
|
||||
|
||||
def last_token(name: str) -> str:
|
||||
parts = name.replace("-", " ").split()
|
||||
return parts[-1] if parts else name
|
||||
|
||||
|
||||
def _split_entries(text: str) -> list[tuple[str, str, str]]:
|
||||
"""Retorna [(type, key, body), ...] respeitando aninhamento de chaves."""
|
||||
entries: list[tuple[str, str, str]] = []
|
||||
i = 0
|
||||
while True:
|
||||
m = re.compile(r"@(\w+)\s*\{").search(text, i)
|
||||
if not m:
|
||||
break
|
||||
etype, key = m.group(1), ""
|
||||
start = m.end()
|
||||
depth = 1
|
||||
j = start
|
||||
while j < len(text) and depth:
|
||||
if text[j] == "{":
|
||||
depth += 1
|
||||
elif text[j] == "}":
|
||||
depth -= 1
|
||||
j += 1
|
||||
body = text[start : j - 1]
|
||||
|
||||
# chave = conteúdo até a primeira vírgula fora de chaves
|
||||
depth = 0
|
||||
k = 0
|
||||
while k < len(body):
|
||||
c = body[k]
|
||||
if c == "{":
|
||||
depth += 1
|
||||
elif c == "}":
|
||||
depth -= 1
|
||||
elif c == "," and depth == 0:
|
||||
break
|
||||
if c == "\n" and depth == 0:
|
||||
break
|
||||
k += 1
|
||||
key = body[:k].strip()
|
||||
entries.append((etype, key, body[k + 1 :]))
|
||||
i = j
|
||||
return entries
|
||||
|
||||
|
||||
def _parse_fields(body: str) -> dict[str, str]:
|
||||
fields: dict[str, str] = {}
|
||||
i = 0
|
||||
n = len(body)
|
||||
while i < n:
|
||||
while i < n and body[i] in " \t\n,":
|
||||
i += 1
|
||||
if i >= n:
|
||||
break
|
||||
m = re.compile(r"(\w+)\s*=").match(body, i)
|
||||
if not m:
|
||||
break
|
||||
fname = m.group(1).lower()
|
||||
i = m.end()
|
||||
while i < n and body[i] in " \t\n":
|
||||
i += 1
|
||||
if i >= n:
|
||||
break
|
||||
if body[i] == "{":
|
||||
depth = 1
|
||||
j = i + 1
|
||||
while j < n and depth:
|
||||
if body[j] == "{":
|
||||
depth += 1
|
||||
elif body[j] == "}":
|
||||
depth -= 1
|
||||
j += 1
|
||||
value = body[i + 1 : j - 1]
|
||||
i = j
|
||||
else:
|
||||
m2 = re.compile(r"([^,]+)").match(body, i)
|
||||
if not m2:
|
||||
i += 1
|
||||
continue
|
||||
value = m2.group(1)
|
||||
i = m2.end()
|
||||
value = value.strip()
|
||||
value = re.sub(r"\s*\n\s*", " ", value)
|
||||
value = value.replace("{", "").replace("}", "").strip()
|
||||
value = _clean_bib_value(value)
|
||||
fields[fname] = value
|
||||
return fields
|
||||
|
||||
|
||||
def _clean_bib_value(value: str) -> str:
|
||||
"""Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.).
|
||||
|
||||
Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``,
|
||||
``Tirke\\cs`` -> ``Tirkes``.
|
||||
"""
|
||||
value = re.sub(
|
||||
r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))",
|
||||
lambda m: m.group(1) or m.group(2),
|
||||
value,
|
||||
flags=re.I,
|
||||
)
|
||||
for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"),
|
||||
("\\_", "_"), ("\\#", "#")):
|
||||
value = value.replace(esc, ch)
|
||||
value = value.replace("---", " — ").replace("--", " – ")
|
||||
return value
|
||||
|
||||
|
||||
def parse_bib(bib_path: Path | str) -> dict[str, Reference]:
|
||||
text = Path(bib_path).read_text(encoding="utf-8")
|
||||
refs: dict[str, Reference] = {}
|
||||
for etype, key, body in _split_entries(text):
|
||||
if not key:
|
||||
continue
|
||||
refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body))
|
||||
return refs
|
||||
172
analise_artigo/report.py
Normal file
172
analise_artigo/report.py
Normal file
@@ -0,0 +1,172 @@
|
||||
"""Gera o relatório em Markdown e o artefato JSON."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from .analyze import Analysis
|
||||
from .bib import Reference
|
||||
from .texparse import count_words
|
||||
|
||||
|
||||
def _md_escape(text: str) -> str:
|
||||
return text.replace("|", "\\|")
|
||||
|
||||
|
||||
def _ref_line(ref: Reference) -> str:
|
||||
parts = [ref.year and ref.year, ref.title, ref.authors, ref.journal]
|
||||
parts = [re.sub(r"\s*\n\s*", " ", p) for p in parts if p]
|
||||
if not parts:
|
||||
return ref.key
|
||||
return "; ".join(parts) if len(parts) > 1 else parts[0]
|
||||
|
||||
|
||||
def build_markdown(a: Analysis) -> str:
|
||||
art, m = a.article, a.metrics
|
||||
lines: list[str] = []
|
||||
add = lines.append
|
||||
|
||||
add(f"# Análise do artigo")
|
||||
add("")
|
||||
add(f"> **{art.title}**")
|
||||
add(f"> Autores: {', '.join(art.authors)} — {art.local}, {art.data} — {art.instituicao}")
|
||||
add("")
|
||||
|
||||
resumo = m["resumo_checks"]
|
||||
add(f"**Resultado global:** ✅ {resumo['ok']} ok · ⚠️ {resumo['warn']} aviso(s) · ❌ {resumo['fail']} falha(s) · 📝 {len(a.issues)} questão(ões) de prosa")
|
||||
add("")
|
||||
|
||||
# 1. Estrutura ---------------------------------------------------------
|
||||
add("## 1. Estrutura e elementos")
|
||||
add("")
|
||||
add("| # | Verificação | Status | Detalhe |")
|
||||
add("|---|---|---|---|")
|
||||
for c in a.checks:
|
||||
add(f"| {c.code} | {c.label} | {c.icon} | {_md_escape(c.detail)} |")
|
||||
add("")
|
||||
|
||||
# 2. Métricas textuais ------------------------------------------------
|
||||
add("## 2. Métricas textuais")
|
||||
add("")
|
||||
add("| Métrica | Valor |")
|
||||
add("|---|---|")
|
||||
add(f"| Palavras (corpo) | {m['palavras_corpo']:,} |")
|
||||
add(f"| Palavras (total, c/ resumo) | {m['palavras_totais']:,} |")
|
||||
add(f"| Frases | {m['sentencas']} |")
|
||||
add(f"| Média de palavras por frase | {m['media_palavras_por_sentenca']} |")
|
||||
add(f"| Vocabulário único | {m['vocabulario_unico']:,} |")
|
||||
add(f"| Legibilidade (Flesch adaptado) | {m['flesch']} — {m['flesch_faixa']} |")
|
||||
add("")
|
||||
|
||||
add("### Distribuição por seção")
|
||||
add("")
|
||||
add("| Seção | Palavras | % do total |")
|
||||
add("|---|---:|---:|")
|
||||
for s in m["secoes"]:
|
||||
add(f"| {s['titulo']} | {s['palavras']:,} | {s['percentual']}% |")
|
||||
add("")
|
||||
|
||||
# 3. Citações ----------------------------------------------------------
|
||||
stats = m["citacoes"]
|
||||
add("## 3. Citações e referências (NBR 10520 / 6023)")
|
||||
add("")
|
||||
add(f"- Citações no texto: **{stats['total_citations']}** ({stats['unique_cited']} referências distintas)")
|
||||
add(f"- Entradas no `.bib`: **{stats['defined_refs']}**")
|
||||
if stats["orphan_citations"]:
|
||||
add(f"- ❌ Citadas mas **ausentes do .bib**: {', '.join(stats['orphan_citations'])}")
|
||||
if stats["unused_refs"]:
|
||||
add(f"- ⚠️ No .bib mas **não citadas no texto**: {', '.join(stats['unused_refs'])}")
|
||||
add("")
|
||||
if stats["most_cited"]:
|
||||
add("### Referências mais citadas")
|
||||
add("")
|
||||
add("| Referência | Cit.ªs | Entrada |")
|
||||
add("|---|---:|---|")
|
||||
for key, n, label in stats["most_cited"]:
|
||||
add(f"| {_md_escape(label)} | {n} | `{key}` |")
|
||||
add("")
|
||||
if stats["year_distribution"]:
|
||||
add("### Referências por ano")
|
||||
add("")
|
||||
years = sorted(stats["year_distribution"].items())
|
||||
add("| Ano | N.º |")
|
||||
add("|---:|---:|")
|
||||
for year, n in years:
|
||||
add(f"| {year} | {n} |")
|
||||
add("")
|
||||
|
||||
# 4. Prosa -------------------------------------------------------------
|
||||
add("## 4. Questões de prosa")
|
||||
add("")
|
||||
if a.issues:
|
||||
for i in a.issues:
|
||||
add(f"- {i}")
|
||||
else:
|
||||
add("Nenhuma questão detectada pelas heurísticas.")
|
||||
add("")
|
||||
if m["top_frases"]:
|
||||
add("### Expressões mais repetidas (5 palavras)")
|
||||
add("")
|
||||
for phrase, n in m["top_frases"]:
|
||||
add(f"- \"{phrase}\" — {n}×")
|
||||
add("")
|
||||
|
||||
# 5. Referências -------------------------------------------------------
|
||||
add("## 5. Referências do .bib")
|
||||
add("")
|
||||
for key in sorted(a.refs):
|
||||
ref = a.refs[key]
|
||||
cited = key in set(a.article.citations)
|
||||
mark = "✅" if cited else "⚠️"
|
||||
add(f"{mark} `{key}` — {_md_escape(_ref_line(ref))}")
|
||||
add("")
|
||||
|
||||
add("---")
|
||||
add(f"_Gerado em {datetime.now(timezone.utc):%Y-%m-%d %H:%M} UTC "
|
||||
f"pelo projeto `analise-artigo` (v1.0)._")
|
||||
return "\n".join(lines) + "\n"
|
||||
|
||||
|
||||
def build_json(a: Analysis) -> str:
|
||||
d = {
|
||||
"article": {
|
||||
"title": a.article.title,
|
||||
"authors": a.article.authors,
|
||||
"local": a.article.local,
|
||||
"data": a.article.data,
|
||||
"instituicao": a.article.instituicao,
|
||||
"keywords": a.article.keywords,
|
||||
"abstract_words": len(count_words(a.article.abstract)),
|
||||
"sections": [
|
||||
{"titulo": s.title, "nivel": s.level, "palavras": s.words}
|
||||
for s in a.article.sections
|
||||
],
|
||||
"figures": a.article.figures,
|
||||
"tables": a.article.tables,
|
||||
},
|
||||
"checks": [
|
||||
{"code": c.code, "label": c.label, "status": c.status,
|
||||
"detail": c.detail}
|
||||
for c in a.checks
|
||||
],
|
||||
"metrics": a.metrics,
|
||||
"issues": a.issues,
|
||||
"references": {
|
||||
k: {"type": r.type, "fields": r.fields,
|
||||
"cited": k in set(a.article.citations)}
|
||||
for k, r in a.refs.items()
|
||||
},
|
||||
}
|
||||
return json.dumps(d, ensure_ascii=False, indent=2)
|
||||
|
||||
|
||||
def write_reports(a: Analysis, outdir: Path) -> tuple[Path, Path]:
|
||||
outdir.mkdir(parents=True, exist_ok=True)
|
||||
md_path = outdir / "relatorio.md"
|
||||
json_path = outdir / "relatorio.json"
|
||||
md_path.write_text(build_markdown(a), encoding="utf-8")
|
||||
json_path.write_text(build_json(a), encoding="utf-8")
|
||||
return md_path, json_path
|
||||
264
analise_artigo/texparse.py
Normal file
264
analise_artigo/texparse.py
Normal file
@@ -0,0 +1,264 @@
|
||||
"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT).
|
||||
|
||||
Converte o .tex em texto limpo preservando:
|
||||
- título, autores, local/data (elementos pré-textuais);
|
||||
- resumo e palavras-chave;
|
||||
- seções/subseções e seus parágrafos;
|
||||
- chaves de citação (\\cite, \\citeonline, \\citeauthor ...).
|
||||
|
||||
Não depende de pandoc: usa apenas a biblioteca padrão.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@dataclass
|
||||
class Section:
|
||||
"""Seção do artigo com o texto limpo já convertido para prosa."""
|
||||
|
||||
title: str
|
||||
level: int = 1
|
||||
|
||||
@property
|
||||
def words(self) -> int:
|
||||
return len(count_words(self.text))
|
||||
|
||||
|
||||
@dataclass
|
||||
class Article:
|
||||
title: str = ""
|
||||
authors: list[str] = field(default_factory=list)
|
||||
local: str = ""
|
||||
data: str = ""
|
||||
instituicao: str = ""
|
||||
abstract: str = ""
|
||||
keywords: list[str] = field(default_factory=list)
|
||||
sections: list[Section] = field(default_factory=list)
|
||||
citations: list[str] = field(default_factory=list) # ordem de aparição
|
||||
figures: list[str] = field(default_factory=list) # captions das figuras
|
||||
tables: list[str] = field(default_factory=list) # captions de quadros
|
||||
text_words: int = 0
|
||||
|
||||
def section(self, title: str) -> Section | None:
|
||||
for s in self.sections:
|
||||
if s.title.strip().lower() == title.strip().lower():
|
||||
return s
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Conversão LaTeX -> texto
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline",
|
||||
"bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"]
|
||||
|
||||
_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}")
|
||||
_BRACE_CMD_RE = {
|
||||
cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS
|
||||
}
|
||||
# Chaves com barra escapada no LaTeX (ex.: "\_", "\#").
|
||||
# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima.
|
||||
_ESCAPE_MAP = {
|
||||
"\\_": "_",
|
||||
"\\#": "#",
|
||||
"\\\\": "\n",
|
||||
}
|
||||
|
||||
|
||||
def count_words(text: str) -> list[str]:
|
||||
return re.findall(
|
||||
r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text
|
||||
)
|
||||
|
||||
|
||||
def _fold_brace_commands(s: str) -> str:
|
||||
changed = True
|
||||
while changed:
|
||||
changed = False
|
||||
for cmd, pat in _BRACE_CMD_RE.items():
|
||||
s2, n = pat.subn(r"\1", s)
|
||||
if n:
|
||||
s, changed = s2, True
|
||||
return s
|
||||
|
||||
|
||||
def strip_latex(text: str, citations: list[str] | None = None) -> str:
|
||||
"""Remove comandos e ambientes, preservando o conteúdo legível."""
|
||||
|
||||
# citações: captura as chaves e remove do texto
|
||||
def _cite(m: re.Match) -> str:
|
||||
if citations is not None:
|
||||
for key in m.group(1).split(","):
|
||||
key = key.strip()
|
||||
if key:
|
||||
citations.append(key)
|
||||
return " "
|
||||
|
||||
text = _CITE_RE.sub(_cite, text)
|
||||
|
||||
# rótulos/referências internas e figuras
|
||||
text = re.sub(r"\\label\{[^}]*\}", " ", text)
|
||||
text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text)
|
||||
text = re.sub(r"\\includegraphics.*", " ", text)
|
||||
text = re.sub(r"\\noindent", " ", text)
|
||||
|
||||
# comandos com argumento em chaves genéricos (legend, caption, textbf...)
|
||||
text = _fold_brace_commands(text)
|
||||
text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text)
|
||||
|
||||
# ambientes: manter marcadores de item, eliminar o resto
|
||||
text = re.sub(r"\\begin\{itemize\}", "\n- ", text)
|
||||
text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text)
|
||||
text = re.sub(r"\\item\b", "\n- ", text)
|
||||
text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text)
|
||||
text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text)
|
||||
text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor
|
||||
|
||||
# comandos restantes sem argumento
|
||||
text = re.sub(r"\\[a-zA-Z]+\*?", " ", text)
|
||||
text = re.sub(r"\\[,;:\\!%$&]", " ", text)
|
||||
for esc, ch in _ESCAPE_MAP.items():
|
||||
text = text.replace(esc, ch)
|
||||
|
||||
# aspas e travessões
|
||||
text = text.replace("``", '"').replace("''", '"')
|
||||
text = text.replace("---", " — ").replace("--", " – ")
|
||||
|
||||
# tabular: & vira coluna
|
||||
text = text.replace("&", " | ")
|
||||
|
||||
# limpar pontuação e espaços
|
||||
text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb]
|
||||
text = re.sub(r"[{}]+", " ", text)
|
||||
text = re.sub(r"[ \t]+", " ", text)
|
||||
text = re.sub(r" +([,.;:!?])", r"\1", text)
|
||||
text = re.sub(r"\n\s*", "\n", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Parsing da estrutura
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}")
|
||||
|
||||
|
||||
def _capture_command(text: str, name: str) -> str:
|
||||
"""Extrai o último conteúdo de \\name{...} ou \\<name> ... \\end{<name>}."""
|
||||
m = re.search(r"\\" + name + r"\s*\{", text)
|
||||
if not m:
|
||||
return ""
|
||||
start = m.end()
|
||||
depth = 1
|
||||
i = start
|
||||
while i < len(text) and depth:
|
||||
if text[i] == "{":
|
||||
depth += 1
|
||||
elif text[i] == "}":
|
||||
depth -= 1
|
||||
i += 1
|
||||
return text[start : i - 1].strip()
|
||||
|
||||
|
||||
def _extract_authors(text: str) -> list[str]:
|
||||
raw = _capture_command(text, "autor")
|
||||
if not raw:
|
||||
raw = _capture_command(text, "author")
|
||||
return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()]
|
||||
|
||||
|
||||
def parse_article(tex_path: Path | str) -> Article:
|
||||
tex_path = Path(tex_path)
|
||||
text = tex_path.read_text(encoding="utf-8")
|
||||
|
||||
# remove comentários de linha (não escapados)
|
||||
text = re.sub(r"(?<!\\)%[^\n]*", "", text)
|
||||
|
||||
citations: list[str] = []
|
||||
raw_sections: list[tuple[int, str, list[str]]] = []
|
||||
|
||||
title = strip_latex(_capture_command(text, "titulo")
|
||||
or _capture_command(text, "title"))
|
||||
authors = _extract_authors(text)
|
||||
local = strip_latex(_capture_command(text, "local"))
|
||||
data = strip_latex(_capture_command(text, "data"))
|
||||
instituicao = strip_latex(_capture_command(text, "instituicao")
|
||||
or _capture_command(text, "institution"))
|
||||
|
||||
# resumo (ambiente resumoumacoluna do abntex2)
|
||||
m = re.search(r"\\begin\{resumoumacoluna\}(.*?)\\end\{resumoumacoluna\}",
|
||||
text, re.S)
|
||||
abstract_raw = m.group(1) if m else ""
|
||||
abstract = strip_latex(abstract_raw, citations)
|
||||
|
||||
keywords: list[str] = []
|
||||
km = re.search(
|
||||
r"Palavras-chave[: ]*\s*(.+)", abstract
|
||||
)
|
||||
if km:
|
||||
kw_body = strip_latex(abstract)
|
||||
kw_text = kw_body.split("Palavras-chave", 1)[-1]
|
||||
kw_text = kw_text.split(":", 1)[-1].strip()
|
||||
keywords = [k.strip() for k in re.split(r"[.;]", kw_text) if k.strip()]
|
||||
|
||||
# seções: percorre linha a linha acumulando o corpo de cada seção
|
||||
current: tuple[int, str, list[str]] | None = None
|
||||
preamble_text = ""
|
||||
figures: list[str] = []
|
||||
tables: list[str] = []
|
||||
|
||||
for line in text.splitlines():
|
||||
sm = _SECTION_RE.match(line)
|
||||
if sm:
|
||||
level = {"section": 1, "subsection": 2, "subsubsection": 3}[sm.group(1)]
|
||||
title_s = strip_latex(sm.group(2), citations)
|
||||
if current:
|
||||
raw_sections.append(current)
|
||||
current = (level, title_s, [])
|
||||
continue
|
||||
|
||||
if current is None:
|
||||
preamble_text += line + "\n"
|
||||
continue
|
||||
current[2].append(line)
|
||||
|
||||
if current:
|
||||
raw_sections.append(current)
|
||||
|
||||
article = Article(
|
||||
title=title,
|
||||
authors=authors,
|
||||
local=local,
|
||||
data=data,
|
||||
instituicao=instituicao,
|
||||
abstract=abstract,
|
||||
keywords=keywords,
|
||||
)
|
||||
|
||||
for level, title_s, lines in raw_sections:
|
||||
sec = Section(title=title_s, level=level)
|
||||
body = "\n".join(lines)
|
||||
# captura captions de figuras/quadros antes de remover
|
||||
for cap in re.findall(r"\\caption\{([^}]+)\}", body):
|
||||
if "\\includegraphics" in body or "figura" in body.lower():
|
||||
figures.append(strip_latex(cap))
|
||||
else:
|
||||
tables.append(strip_latex(cap))
|
||||
sec.text = strip_latex(body, citations)
|
||||
article.sections.append(sec)
|
||||
|
||||
article.citations = citations
|
||||
article.figures = figures
|
||||
article.tables = tables
|
||||
article.text_words = (
|
||||
len(count_words(title))
|
||||
+ len(count_words(abstract))
|
||||
+ sum(len(count_words(s.text)) for s in article.sections)
|
||||
)
|
||||
return article
|
||||
Reference in New Issue
Block a user