Projeto analise-artigo: análise ABNT de artigos LaTeX

- texparse.py: extração de estrutura abntex2 (título, autores, resumo,
  palavras-chave, seções, citações, figuras), sem dependências externas
- bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos
- analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch
  adaptado pt-BR e heurísticas de prosa
- report.py: relatório Markdown + JSON
- artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
This commit is contained in:
2026-09-25 13:24:56 -03:00
commit 550440f4a7
14 changed files with 1591 additions and 0 deletions

View File

@@ -0,0 +1,23 @@
"""Análise de artigos acadêmicos em LaTeX (ABNT/abntex2).
Empacota os módulos de parsing (LaTeX e .bib), métricas,
verificações e geração de relatório.
"""
from .texparse import Article, parse_article
from .bib import Reference, parse_bib
from .analyze import Analysis, analyze
from .report import build_markdown, build_json
__all__ = [
"Article",
"parse_article",
"Reference",
"parse_bib",
"Analysis",
"analyze",
"build_markdown",
"build_json",
]
__version__ = "1.0.0"

View File

@@ -0,0 +1,85 @@
"""Linha de comando.
Exemplos:
python -m analise_artigo artigos/rumo-a-eficiencia --out relatorios
python -m analise_artigo caminho/para/main.tex
"""
from __future__ import annotations
import argparse
import sys
from pathlib import Path
from .bib import parse_bib
from .report import write_reports
from .texparse import parse_article
from .analyze import analyze
def find_tex(target: Path) -> Path | None:
if target.is_file() and target.suffix == ".tex":
return target
if target.is_dir():
candidates = sorted(target.glob("*.tex"))
return candidates[0] if candidates else None
return None
def find_bib(tex: Path) -> Path | None:
bib = tex.with_name(tex.stem + ".bib")
if bib.exists():
return bib
for candidate in sorted(tex.parent.glob("*.bib")):
return candidate
return None
def main(argv: list[str] | None = None) -> int:
p = argparse.ArgumentParser(
prog="analise-artigo",
description="Analisa um artigo LaTeX (ABNT) e gera relatório de "
"estrutura, métricas e citações.",
)
p.add_argument(
"artigo",
help="arquivo .tex ou diretório que contém o .tex",
)
p.add_argument(
"--out", "-o",
default="relatorios",
help="diretório de saída do relatório (padrão: relatorios/)",
)
args = p.parse_args(argv)
target = Path(args.artigo)
tex = find_tex(target)
if tex is None:
print(f"erro: nenhum .tex encontrado em {target}", file=sys.stderr)
return 1
bib = find_bib(tex)
print(f"→ artigo: {tex}")
print(f"→ referências: {bib or 'nenhum .bib encontrado'}")
article = parse_article(tex)
refs = parse_bib(bib) if bib else {}
analysis = analyze(article, refs)
outdir = Path(args.out)
md, js = write_reports(analysis, outdir)
resumo = analysis.metrics["resumo_checks"]
print(f"\nResumo: {resumo['ok']} ok, "
f"{resumo['warn']} aviso(s), "
f"{resumo['fail']} falha(s), "
f"{len(analysis.issues)} questão(ões) de prosa")
print(f"\n→ relatório: {md}")
print(f"→ dados: {js}")
return 0 if resumo["fail"] == 0 else 2
if __name__ == "__main__":
raise SystemExit(main())

346
analise_artigo/analyze.py Normal file
View File

@@ -0,0 +1,346 @@
"""Métricas textuais, consistência de citações e checklist de conformidade.
Regras ABNT/NBR usadas:
- NBR 10520: citações no texto devem existir na referências e vice-versa;
- NBR 6022: elementos obrigatórios do artigo (título, resumo,
palavras-chave, seções, referências);
- NBR 14724: elementos pré/textuais/pós-textuais.
"""
from __future__ import annotations
import re
from collections import Counter
from dataclasses import dataclass, field
from .bib import Reference
from .texparse import Article, Section, count_words
# ---------------------------------------------------------------------------
# Estruturas de resultado
# ---------------------------------------------------------------------------
@dataclass
class Check:
code: str
label: str
status: str # ok | warn | fail
detail: str = ""
@property
def icon(self) -> str:
return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status]
@dataclass
class Analysis:
article: Article
refs: dict[str, Reference]
checks: list[Check] = field(default_factory=list)
metrics: dict = field(default_factory=dict)
issues: list[str] = field(default_factory=list)
# ---------------------------------------------------------------------------
# Métricas de prosa
# ---------------------------------------------------------------------------
VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]"
def syllables(word: str) -> int:
"""Aproximação de sílabas: 1 sílaba por grupo de vogais.
Contar *grupos* (em vez de caracteres) evita inflar ditongos como
"ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto).
"""
w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower())
if not w:
return 1
n = len(re.findall(VOWELS + "+", w))
return max(1, n)
def sentences(text: str) -> list[str]:
"""Divide em frases respeitando as quebras de linha do .tex.
Nos artigos processados cada parágrafo/item de lista ocupa uma linha
própria (convenção verificada nas fontes), então linha vira fronteira
segura de frase; dentro da linha, pontuação [.!?] segue valendo.
"""
out: list[str] = []
for line in text.splitlines():
line = line.strip()
if not line:
continue
parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line)
for p in parts:
p = p.strip()
# ignora marcadores de lista ("-", "1.") e ruído trivial
if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p):
continue
out.append(p)
return out
def flesch(text: str) -> float | None:
"""Índice de Flesch adaptado para o português (0-100; mais alto = mais legível).
Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com
sílabas estimadas por grupos de vogais (aproximação, sem acentuação
completa). Resultado clamped em [0, 100].
"""
words = count_words(text)
sents = sentences(text)
if not words or not sents:
return None
syl = sum(syllables(w) for w in words)
mpf = len(words) / len(sents) # média de palavras por frase
vmp = syl / len(words) # vocabulário médio (sílabas/palavra)
score = 206.835 - 1.015 * mpf - 84.6 * vmp
return round(min(100.0, max(0.0, score)), 1)
def flesch_band(score: float | None) -> str:
if score is None:
return "n/d"
if score >= 80:
return "Muito fácil"
if score >= 60:
return "Fácil"
if score >= 40:
return "Médio"
if score >= 20:
return "Difícil"
return "Muito difícil"
# ---------------------------------------------------------------------------
# Heurísticas de qualidade
# ---------------------------------------------------------------------------
REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I)
REDUNDANT = re.compile(
r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I
)
LONG_SENTENCE = 60 # palavras
LONG_PARAGRAPH = 160 # palavras
def find_grammar_issues(text: str) -> list[str]:
issues: list[str] = []
for m in REPEATED_WORD.finditer(text):
# "se se" é legítimo em português pronominal ("avaliou-se se cada
# estudo..."); filtrado para não gerar falso-positivo
if m.group(1).lower() == "se":
continue
issues.append(
f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — "
f"...{context_around(text, m.start())}..."
)
for m in REDUNDANT.finditer(text):
issues.append(
f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — "
f"...{context_around(text, m.start())}..."
)
return issues
def context_around(text: str, pos: int, span: int = 60) -> str:
start = max(0, pos - span)
end = min(len(text), pos + span)
snippet = text[start:end].replace("\n", " ")
return ("…" + snippet) if start > 0 else snippet
def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]:
out = []
for s in sentences(text):
n = len(count_words(s))
if n > limit:
out.append((s[:120] + ("…" if len(s) > 120 else ""), n))
return sorted(out, key=lambda x: -x[1])[:10]
def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]:
# Convenção das fontes analisadas: um parágrafo por linha no .tex,
# portanto cada linha (depois da limpeza) é candidata a parágrafo.
out = []
for p in (line.strip() for line in text.splitlines()):
if not p:
continue
n = len(count_words(p))
if n > limit:
out.append((p[:120] + "…", n))
return sorted(out, key=lambda x: -x[1])[:10]
def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]:
words = [w.lower() for w in count_words(text)]
stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os",
"as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são",
"and", "or"} # operadores de busca do estilo PRISMA (AND/OR)
words = [w for w in words if w not in stop and len(w) > 2]
grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)]
return Counter(grams).most_common(top)
def citation_stats(article: Article, refs: dict[str, Reference]) -> dict:
cited = Counter(article.citations)
defined = set(refs)
used = set(cited)
return {
"total_citations": len(article.citations),
"unique_cited": len(used),
"defined_refs": len(defined),
"orphan_citations": sorted(used - defined),
"unused_refs": sorted(defined - used),
"most_cited": [
(k, n, refs.get(k).label() if k in refs else "?")
for k, n in cited.most_common(8)
],
"year_distribution": dict(
Counter(
(refs[k].year or "?") if k in refs else "?" for k in used
)
),
}
# ---------------------------------------------------------------------------
# Análise completa
# ---------------------------------------------------------------------------
RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"]
EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura",
"estado da arte", "revisão de literatura",
"fundamentação", "referencial"]
def analyze(article: Article, refs: dict[str, Reference]) -> Analysis:
checks: list[Check] = []
issues: list[str] = []
# --- estrutura / pré-textuais ----------------------------------------
checks.append(Check(
"ABNT-01", "Título presente",
"ok" if article.title else "fail",
article.title or "não encontrado no \\titulo{}",
))
checks.append(Check(
"ABNT-02", "Autores informados",
"ok" if article.authors else "fail",
", ".join(article.authors) or "não encontrado",
))
checks.append(Check(
"ABNT-03", "Local e data",
"ok" if article.local and article.data else "warn",
f"{article.local}, {article.data}".strip(",") or "incompleto",
))
abs_words = len(count_words(article.abstract))
checks.append(Check(
"ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)",
"ok" if 150 <= abs_words <= 500 else "warn",
f"{abs_words} palavras" if abs_words else "não encontrado",
))
checks.append(Check(
"ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)",
"ok" if 3 <= len(article.keywords) <= 8 else "warn",
", ".join(article.keywords) or "não encontradas",
))
all_titles = [s.title.lower() for s in article.sections]
for expected in RECOGNIZED_SECTIONS:
hit = next((t for t in all_titles if expected in t), None)
checks.append(Check(
"ABNT-06", f"Seção '{expected.capitalize()}'",
"ok" if hit else "warn",
hit or "não identificada",
))
lit_hit = next((t for t in all_titles
if any(e in t for e in EXPECTED_LITERATURE)), None)
checks.append(Check(
"ABNT-07", "Seção de revisão de literatura",
"ok" if lit_hit else "fail",
lit_hit or "faltando",
))
# citações / referências ------------------------------------------------
stats = citation_stats(article, refs)
checks.append(Check(
"ABNT-08", "Citações no corpo do texto",
"ok" if stats["total_citations"] >= 5 else "warn",
f"{stats['total_citations']} citações, "
f"{stats['unique_cited']} referências citadas",
))
checks.append(Check(
"ABNT-09", "Citações sem entrada no .bib (NBR 10520)",
"ok" if not stats["orphan_citations"] else "fail",
", ".join(stats["orphan_citations"]) or "nenhuma",
))
checks.append(Check(
"ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)",
"ok" if not stats["unused_refs"] else "warn",
", ".join(stats["unused_refs"]) or "nenhuma",
))
# figuras ------------------------------------------------------------
checks.append(Check(
"ABNT-11", "Figuras/quadros identificados",
"ok" if (article.figures or article.tables) else "warn",
f"{len(article.figures)} figuras, {len(article.tables)} quadros",
))
# prosa ---------------------------------------------------------------
body_text = "\n\n".join(s.text for s in article.sections)
flesch_score = flesch(body_text)
top_phrases = frequent_phrases(body_text)
repeated = find_grammar_issues(article.abstract + " " + body_text)
l_sentences = long_sentences(body_text)
l_paragraphs = long_paragraphs(body_text)
for i in repeated:
issues.append(i)
for s, n in l_sentences:
issues.append(f"Frase longa ({n} palavras): \"{s}\"")
for p, n in l_paragraphs:
issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"")
metrics = {
"palavras_corpo": len(count_words(body_text)),
"palavras_totais": article.text_words,
"sentencas": len(sentences(body_text)),
"media_palavras_por_sentenca": round(
len(count_words(body_text)) / max(1, len(sentences(body_text))), 1
),
"vocabulario_unico": len(
set(w.lower() for w in count_words(body_text))
),
"flesch": flesch_score,
"flesch_faixa": flesch_band(flesch_score),
"secoes": [
{
"titulo": s.title,
"nivel": s.level,
"palavras": s.words,
"percentual": round(
100 * s.words / max(1, article.text_words), 1
),
}
for s in article.sections
],
"top_frases": top_phrases,
"citacoes": stats,
}
summary = {
status: sum(1 for c in checks if c.status == status)
for status in ("ok", "warn", "fail")
}
metrics["resumo_checks"] = summary
return Analysis(
article=article, refs=refs, checks=checks, metrics=metrics,
issues=issues,
)

167
analise_artigo/bib.py Normal file
View File

@@ -0,0 +1,167 @@
"""Parser simples de arquivos .bib (BibTeX).
Usa apenas a biblioteca padrão; suporta campos com valores em chaves
aninhadas e valores multi-linha.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Reference:
key: str
type: str # article, inproceedings, misc, ...
fields: dict[str, str] = field(default_factory=dict)
@property
def authors(self) -> str:
value = self.fields.get("author", "").strip()
# "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author)
value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I)
return re.sub(r"\s+and\s+", ", ", value, flags=re.I)
@property
def title(self) -> str:
return self.fields.get("title", "").strip()
@property
def year(self) -> str:
return self.fields.get("year", "").strip()
@property
def journal(self) -> str:
return self.fields.get(
"journal", self.fields.get("booktitle", "")
).strip()
def label(self) -> str:
"""Rótulo curto autor (ano)."""
first = re.split(r",|and", self.authors)[0].strip()
if "," in first:
surname = first.split(",")[0].strip()
else:
surname = last_token(first)
return f"{surname} ({self.year})" if self.year else surname
def last_token(name: str) -> str:
parts = name.replace("-", " ").split()
return parts[-1] if parts else name
def _split_entries(text: str) -> list[tuple[str, str, str]]:
"""Retorna [(type, key, body), ...] respeitando aninhamento de chaves."""
entries: list[tuple[str, str, str]] = []
i = 0
while True:
m = re.compile(r"@(\w+)\s*\{").search(text, i)
if not m:
break
etype, key = m.group(1), ""
start = m.end()
depth = 1
j = start
while j < len(text) and depth:
if text[j] == "{":
depth += 1
elif text[j] == "}":
depth -= 1
j += 1
body = text[start : j - 1]
# chave = conteúdo até a primeira vírgula fora de chaves
depth = 0
k = 0
while k < len(body):
c = body[k]
if c == "{":
depth += 1
elif c == "}":
depth -= 1
elif c == "," and depth == 0:
break
if c == "\n" and depth == 0:
break
k += 1
key = body[:k].strip()
entries.append((etype, key, body[k + 1 :]))
i = j
return entries
def _parse_fields(body: str) -> dict[str, str]:
fields: dict[str, str] = {}
i = 0
n = len(body)
while i < n:
while i < n and body[i] in " \t\n,":
i += 1
if i >= n:
break
m = re.compile(r"(\w+)\s*=").match(body, i)
if not m:
break
fname = m.group(1).lower()
i = m.end()
while i < n and body[i] in " \t\n":
i += 1
if i >= n:
break
if body[i] == "{":
depth = 1
j = i + 1
while j < n and depth:
if body[j] == "{":
depth += 1
elif body[j] == "}":
depth -= 1
j += 1
value = body[i + 1 : j - 1]
i = j
else:
m2 = re.compile(r"([^,]+)").match(body, i)
if not m2:
i += 1
continue
value = m2.group(1)
i = m2.end()
value = value.strip()
value = re.sub(r"\s*\n\s*", " ", value)
value = value.replace("{", "").replace("}", "").strip()
value = _clean_bib_value(value)
fields[fname] = value
return fields
def _clean_bib_value(value: str) -> str:
"""Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.).
Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``,
``Tirke\\cs`` -> ``Tirkes``.
"""
value = re.sub(
r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))",
lambda m: m.group(1) or m.group(2),
value,
flags=re.I,
)
for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"),
("\\_", "_"), ("\\#", "#")):
value = value.replace(esc, ch)
value = value.replace("---", " — ").replace("--", " – ")
return value
def parse_bib(bib_path: Path | str) -> dict[str, Reference]:
text = Path(bib_path).read_text(encoding="utf-8")
refs: dict[str, Reference] = {}
for etype, key, body in _split_entries(text):
if not key:
continue
refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body))
return refs

172
analise_artigo/report.py Normal file
View File

@@ -0,0 +1,172 @@
"""Gera o relatório em Markdown e o artefato JSON."""
from __future__ import annotations
import json
import re
from datetime import datetime, timezone
from pathlib import Path
from .analyze import Analysis
from .bib import Reference
from .texparse import count_words
def _md_escape(text: str) -> str:
return text.replace("|", "\\|")
def _ref_line(ref: Reference) -> str:
parts = [ref.year and ref.year, ref.title, ref.authors, ref.journal]
parts = [re.sub(r"\s*\n\s*", " ", p) for p in parts if p]
if not parts:
return ref.key
return "; ".join(parts) if len(parts) > 1 else parts[0]
def build_markdown(a: Analysis) -> str:
art, m = a.article, a.metrics
lines: list[str] = []
add = lines.append
add(f"# Análise do artigo")
add("")
add(f"> **{art.title}**")
add(f"> Autores: {', '.join(art.authors)} — {art.local}, {art.data} — {art.instituicao}")
add("")
resumo = m["resumo_checks"]
add(f"**Resultado global:** ✅ {resumo['ok']} ok · ⚠️ {resumo['warn']} aviso(s) · ❌ {resumo['fail']} falha(s) · 📝 {len(a.issues)} questão(ões) de prosa")
add("")
# 1. Estrutura ---------------------------------------------------------
add("## 1. Estrutura e elementos")
add("")
add("| # | Verificação | Status | Detalhe |")
add("|---|---|---|---|")
for c in a.checks:
add(f"| {c.code} | {c.label} | {c.icon} | {_md_escape(c.detail)} |")
add("")
# 2. Métricas textuais ------------------------------------------------
add("## 2. Métricas textuais")
add("")
add("| Métrica | Valor |")
add("|---|---|")
add(f"| Palavras (corpo) | {m['palavras_corpo']:,} |")
add(f"| Palavras (total, c/ resumo) | {m['palavras_totais']:,} |")
add(f"| Frases | {m['sentencas']} |")
add(f"| Média de palavras por frase | {m['media_palavras_por_sentenca']} |")
add(f"| Vocabulário único | {m['vocabulario_unico']:,} |")
add(f"| Legibilidade (Flesch adaptado) | {m['flesch']} — {m['flesch_faixa']} |")
add("")
add("### Distribuição por seção")
add("")
add("| Seção | Palavras | % do total |")
add("|---|---:|---:|")
for s in m["secoes"]:
add(f"| {s['titulo']} | {s['palavras']:,} | {s['percentual']}% |")
add("")
# 3. Citações ----------------------------------------------------------
stats = m["citacoes"]
add("## 3. Citações e referências (NBR 10520 / 6023)")
add("")
add(f"- Citações no texto: **{stats['total_citations']}** ({stats['unique_cited']} referências distintas)")
add(f"- Entradas no `.bib`: **{stats['defined_refs']}**")
if stats["orphan_citations"]:
add(f"- ❌ Citadas mas **ausentes do .bib**: {', '.join(stats['orphan_citations'])}")
if stats["unused_refs"]:
add(f"- ⚠️ No .bib mas **não citadas no texto**: {', '.join(stats['unused_refs'])}")
add("")
if stats["most_cited"]:
add("### Referências mais citadas")
add("")
add("| Referência | Cit.ªs | Entrada |")
add("|---|---:|---|")
for key, n, label in stats["most_cited"]:
add(f"| {_md_escape(label)} | {n} | `{key}` |")
add("")
if stats["year_distribution"]:
add("### Referências por ano")
add("")
years = sorted(stats["year_distribution"].items())
add("| Ano | N.º |")
add("|---:|---:|")
for year, n in years:
add(f"| {year} | {n} |")
add("")
# 4. Prosa -------------------------------------------------------------
add("## 4. Questões de prosa")
add("")
if a.issues:
for i in a.issues:
add(f"- {i}")
else:
add("Nenhuma questão detectada pelas heurísticas.")
add("")
if m["top_frases"]:
add("### Expressões mais repetidas (5 palavras)")
add("")
for phrase, n in m["top_frases"]:
add(f"- \"{phrase}\" — {n}×")
add("")
# 5. Referências -------------------------------------------------------
add("## 5. Referências do .bib")
add("")
for key in sorted(a.refs):
ref = a.refs[key]
cited = key in set(a.article.citations)
mark = "✅" if cited else "⚠️"
add(f"{mark} `{key}` — {_md_escape(_ref_line(ref))}")
add("")
add("---")
add(f"_Gerado em {datetime.now(timezone.utc):%Y-%m-%d %H:%M} UTC "
f"pelo projeto `analise-artigo` (v1.0)._")
return "\n".join(lines) + "\n"
def build_json(a: Analysis) -> str:
d = {
"article": {
"title": a.article.title,
"authors": a.article.authors,
"local": a.article.local,
"data": a.article.data,
"instituicao": a.article.instituicao,
"keywords": a.article.keywords,
"abstract_words": len(count_words(a.article.abstract)),
"sections": [
{"titulo": s.title, "nivel": s.level, "palavras": s.words}
for s in a.article.sections
],
"figures": a.article.figures,
"tables": a.article.tables,
},
"checks": [
{"code": c.code, "label": c.label, "status": c.status,
"detail": c.detail}
for c in a.checks
],
"metrics": a.metrics,
"issues": a.issues,
"references": {
k: {"type": r.type, "fields": r.fields,
"cited": k in set(a.article.citations)}
for k, r in a.refs.items()
},
}
return json.dumps(d, ensure_ascii=False, indent=2)
def write_reports(a: Analysis, outdir: Path) -> tuple[Path, Path]:
outdir.mkdir(parents=True, exist_ok=True)
md_path = outdir / "relatorio.md"
json_path = outdir / "relatorio.json"
md_path.write_text(build_markdown(a), encoding="utf-8")
json_path.write_text(build_json(a), encoding="utf-8")
return md_path, json_path

264
analise_artigo/texparse.py Normal file
View File

@@ -0,0 +1,264 @@
"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT).
Converte o .tex em texto limpo preservando:
- título, autores, local/data (elementos pré-textuais);
- resumo e palavras-chave;
- seções/subseções e seus parágrafos;
- chaves de citação (\\cite, \\citeonline, \\citeauthor ...).
Não depende de pandoc: usa apenas a biblioteca padrão.
"""
from __future__ import annotations
import re
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Section:
"""Seção do artigo com o texto limpo já convertido para prosa."""
title: str
level: int = 1
@property
def words(self) -> int:
return len(count_words(self.text))
@dataclass
class Article:
title: str = ""
authors: list[str] = field(default_factory=list)
local: str = ""
data: str = ""
instituicao: str = ""
abstract: str = ""
keywords: list[str] = field(default_factory=list)
sections: list[Section] = field(default_factory=list)
citations: list[str] = field(default_factory=list) # ordem de aparição
figures: list[str] = field(default_factory=list) # captions das figuras
tables: list[str] = field(default_factory=list) # captions de quadros
text_words: int = 0
def section(self, title: str) -> Section | None:
for s in self.sections:
if s.title.strip().lower() == title.strip().lower():
return s
return None
# ---------------------------------------------------------------------------
# Conversão LaTeX -> texto
# ---------------------------------------------------------------------------
_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline",
"bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"]
_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}")
_BRACE_CMD_RE = {
cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS
}
# Chaves com barra escapada no LaTeX (ex.: "\_", "\#").
# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima.
_ESCAPE_MAP = {
"\\_": "_",
"\\#": "#",
"\\\\": "\n",
}
def count_words(text: str) -> list[str]:
return re.findall(
r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text
)
def _fold_brace_commands(s: str) -> str:
changed = True
while changed:
changed = False
for cmd, pat in _BRACE_CMD_RE.items():
s2, n = pat.subn(r"\1", s)
if n:
s, changed = s2, True
return s
def strip_latex(text: str, citations: list[str] | None = None) -> str:
"""Remove comandos e ambientes, preservando o conteúdo legível."""
# citações: captura as chaves e remove do texto
def _cite(m: re.Match) -> str:
if citations is not None:
for key in m.group(1).split(","):
key = key.strip()
if key:
citations.append(key)
return " "
text = _CITE_RE.sub(_cite, text)
# rótulos/referências internas e figuras
text = re.sub(r"\\label\{[^}]*\}", " ", text)
text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text)
text = re.sub(r"\\includegraphics.*", " ", text)
text = re.sub(r"\\noindent", " ", text)
# comandos com argumento em chaves genéricos (legend, caption, textbf...)
text = _fold_brace_commands(text)
text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text)
# ambientes: manter marcadores de item, eliminar o resto
text = re.sub(r"\\begin\{itemize\}", "\n- ", text)
text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text)
text = re.sub(r"\\item\b", "\n- ", text)
text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text)
text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text)
text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor
# comandos restantes sem argumento
text = re.sub(r"\\[a-zA-Z]+\*?", " ", text)
text = re.sub(r"\\[,;:\\!%$&]", " ", text)
for esc, ch in _ESCAPE_MAP.items():
text = text.replace(esc, ch)
# aspas e travessões
text = text.replace("``", '"').replace("''", '"')
text = text.replace("---", " — ").replace("--", " – ")
# tabular: & vira coluna
text = text.replace("&", " | ")
# limpar pontuação e espaços
text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb]
text = re.sub(r"[{}]+", " ", text)
text = re.sub(r"[ \t]+", " ", text)
text = re.sub(r" +([,.;:!?])", r"\1", text)
text = re.sub(r"\n\s*", "\n", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
# ---------------------------------------------------------------------------
# Parsing da estrutura
# ---------------------------------------------------------------------------
_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}")
def _capture_command(text: str, name: str) -> str:
"""Extrai o último conteúdo de \\name{...} ou \\<name> ... \\end{<name>}."""
m = re.search(r"\\" + name + r"\s*\{", text)
if not m:
return ""
start = m.end()
depth = 1
i = start
while i < len(text) and depth:
if text[i] == "{":
depth += 1
elif text[i] == "}":
depth -= 1
i += 1
return text[start : i - 1].strip()
def _extract_authors(text: str) -> list[str]:
raw = _capture_command(text, "autor")
if not raw:
raw = _capture_command(text, "author")
return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()]
def parse_article(tex_path: Path | str) -> Article:
tex_path = Path(tex_path)
text = tex_path.read_text(encoding="utf-8")
# remove comentários de linha (não escapados)
text = re.sub(r"(?<!\\)%[^\n]*", "", text)
citations: list[str] = []
raw_sections: list[tuple[int, str, list[str]]] = []
title = strip_latex(_capture_command(text, "titulo")
or _capture_command(text, "title"))
authors = _extract_authors(text)
local = strip_latex(_capture_command(text, "local"))
data = strip_latex(_capture_command(text, "data"))
instituicao = strip_latex(_capture_command(text, "instituicao")
or _capture_command(text, "institution"))
# resumo (ambiente resumoumacoluna do abntex2)
m = re.search(r"\\begin\{resumoumacoluna\}(.*?)\\end\{resumoumacoluna\}",
text, re.S)
abstract_raw = m.group(1) if m else ""
abstract = strip_latex(abstract_raw, citations)
keywords: list[str] = []
km = re.search(
r"Palavras-chave[: ]*\s*(.+)", abstract
)
if km:
kw_body = strip_latex(abstract)
kw_text = kw_body.split("Palavras-chave", 1)[-1]
kw_text = kw_text.split(":", 1)[-1].strip()
keywords = [k.strip() for k in re.split(r"[.;]", kw_text) if k.strip()]
# seções: percorre linha a linha acumulando o corpo de cada seção
current: tuple[int, str, list[str]] | None = None
preamble_text = ""
figures: list[str] = []
tables: list[str] = []
for line in text.splitlines():
sm = _SECTION_RE.match(line)
if sm:
level = {"section": 1, "subsection": 2, "subsubsection": 3}[sm.group(1)]
title_s = strip_latex(sm.group(2), citations)
if current:
raw_sections.append(current)
current = (level, title_s, [])
continue
if current is None:
preamble_text += line + "\n"
continue
current[2].append(line)
if current:
raw_sections.append(current)
article = Article(
title=title,
authors=authors,
local=local,
data=data,
instituicao=instituicao,
abstract=abstract,
keywords=keywords,
)
for level, title_s, lines in raw_sections:
sec = Section(title=title_s, level=level)
body = "\n".join(lines)
# captura captions de figuras/quadros antes de remover
for cap in re.findall(r"\\caption\{([^}]+)\}", body):
if "\\includegraphics" in body or "figura" in body.lower():
figures.append(strip_latex(cap))
else:
tables.append(strip_latex(cap))
sec.text = strip_latex(body, citations)
article.sections.append(sec)
article.citations = citations
article.figures = figures
article.tables = tables
article.text_words = (
len(count_words(title))
+ len(count_words(abstract))
+ sum(len(count_words(s.text)) for s in article.sections)
)
return article