commit 550440f4a7180017289eca5896436d9c01ae3b03 Author: Jário José Date: Fri Sep 25 13:24:56 2026 -0300 Projeto analise-artigo: análise ABNT de artigos LaTeX - texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib) diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..5a0373a --- /dev/null +++ b/.gitignore @@ -0,0 +1,5 @@ +__pycache__/ +*.pyc +.pytest_cache/ +relatorios/ +.venv/ diff --git a/README.md b/README.md new file mode 100644 index 0000000..539fccf --- /dev/null +++ b/README.md @@ -0,0 +1,109 @@ +# analise-artigo + +Projeto de análise de artigos acadêmicos escritos em **LaTeX (ABNT/abntex2)**. +É 100% Python puro (biblioteca padrão), sem dependências externas — roda em +qualquer máquina com Python ≥ 3.10. + +## O que ele analisa + +- **Estrutura e conformidade ABNT** (NBR 10520 / 6022 / 14724) + - título, autores, local/data, resumo (150–500 palavras), palavras-chave + - presença de seções: Introdução, Metodologia, Revisão de Literatura, Conclusão +- **Citações e referências** + - citações no corpo vs. entradas do `.bib` (detecta órfãs e não citadas) + - referências mais citadas e distribuição por ano +- **Métricas textuais** + - palavras, frases, média de tamanho de frase, vocabulário único + - legibilidade (Flesch adaptado ao português) + - distribuição de palavras por seção +- **Qualidade de prosa (heurísticas)** + - palavras repetidas / redundâncias (`de de`, `que que`, `mas mas` …) + - frases > 60 palavras e parágrafos > 160 palavras + - expressões mais repetidas no corpo do artigo + +## Como usar + +```bash +# a partir da raiz do projeto +python -m analise_artigo artigos/rumo-a-eficiencia + +# ou apontando direto para o .tex (o .bib é descoberto automaticamente) +python -m analise_artigo artigos/rumo-a-eficiencia/main.tex + +# saída em outro diretório +python -m analise_artigo caminho/do/artigo --out relatorios/ +``` + +Saídas geradas no diretório de saída (padrão `relatorios/`): + +| Arquivo | Conteúdo | +|-----------------|---------------------------------------------------| +| `relatorio.md` | relatório legível em Markdown | +| `relatorio.json`| dados estruturados (para plotar/automatizar) | + +Código de saída do processo: `0` quando não há verificações em `fail`, +`2` quando há, `1` em erro de uso/entrada. + +## Estrutura do projeto + +``` +analise-artigo/ +├── analise_artigo/ +│ ├── __init__.py # API pública do pacote +│ ├── texparse.py # LaTeX → texto (seções, resumo, citações) +│ ├── bib.py # parser .bib (BibTeX), sem dependências +│ ├── analyze.py # métricas + checklist ABNT + heurísticas +│ ├── report.py # geração do .md e do .json +│ └── __main__.py # CLI +├── artigos/ +│ └── rumo-a-eficiencia/ # o artigo analisado (main.tex + referencias.bib) +├── relatorios/ # saída gerada +├── requirements.txt +└── README.md +``` + +## Usando como biblioteca + +```python +from analise_artigo import parse_article, parse_bib, analyze, write_reports +from pathlib import Path + +artigo = parse_article("artigos/rumo-a-eficiencia/main.tex") +refs = parse_bib("artigos/rumo-a-eficiencia/referencias.bib") +an = analyze(artigo, refs) + +print(an.metrics["flesch"], an.metrics["flesch_faixa"]) +for c in an.checks: + print(c.icon, c.code, c.detail) +``` + +## Limitações conhecidas + +- O parser é orientado a abntex2/ABNT; documentos em outras classes + (`article`, `report` com nomenclatura própria) exigem pequenas adaptações + em `texparse.py`. +- **Uma frase/parágrafo por linha**: as heurísticas de sentença/travessão + e de "parágrafo extenso" presumem que cada parágrafo do .tex ocupe uma + linha (convenção verificada nas fontes analisadas). Textos com parágrafos + múltiplos de linha tendem a ser fragmentados. +- **Flesch adaptado (pt-BR)**: usa a fórmula 206,835 − 1,015·MPF − 84,6·VMP + com sílabas estimadas por grupos de vogais (resultado clamped em 0–100). + Texto acadêmico técnico costuma pontuar baixo (p. ex. 0–15) — o índice + reflete vocabulário denso, não um "erro". +- **"se se" excluído** da detecção de palavras repetidas: é uso pronominal + legítimo ("avaliou-se se cada estudo…"). +- **Simplificação de `.bib`**: nomes com escapes LaTeX recebem diacríticos + removidos para exibição (`Ki\vs\vs\, M.` → `Kiss`), apenas na apresentação + — o arquivo original não é modificado. +- Heurísticas de prosa não substituem revisão humana ou ferramentas como + LanguageTool. + +## Ideias de evolução + +- [ ] Exportar relatório em HTML com gráficos (referências por ano, + palavras por seção) a partir do `relatorio.json` +- [ ] Suporte a bibliografia biblatex (`.bbx`) e listas de referências + manualmente digitadas no corpo do .tex +- [ ] Detecção de consistência terminológica (ex.: "agentes pedagógicos" + vs. "agente pedagógico") +- [ ] Integração com LanguageTool via CLI para correção gramatical completa diff --git a/analise_artigo/__init__.py b/analise_artigo/__init__.py new file mode 100644 index 0000000..f726666 --- /dev/null +++ b/analise_artigo/__init__.py @@ -0,0 +1,23 @@ +"""Análise de artigos acadêmicos em LaTeX (ABNT/abntex2). + +Empacota os módulos de parsing (LaTeX e .bib), métricas, +verificações e geração de relatório. +""" + +from .texparse import Article, parse_article +from .bib import Reference, parse_bib +from .analyze import Analysis, analyze +from .report import build_markdown, build_json + +__all__ = [ + "Article", + "parse_article", + "Reference", + "parse_bib", + "Analysis", + "analyze", + "build_markdown", + "build_json", +] + +__version__ = "1.0.0" diff --git a/analise_artigo/__main__.py b/analise_artigo/__main__.py new file mode 100644 index 0000000..a6f173c --- /dev/null +++ b/analise_artigo/__main__.py @@ -0,0 +1,85 @@ +"""Linha de comando. + +Exemplos: + python -m analise_artigo artigos/rumo-a-eficiencia --out relatorios + python -m analise_artigo caminho/para/main.tex +""" + +from __future__ import annotations + +import argparse +import sys +from pathlib import Path + +from .bib import parse_bib +from .report import write_reports +from .texparse import parse_article +from .analyze import analyze + + +def find_tex(target: Path) -> Path | None: + if target.is_file() and target.suffix == ".tex": + return target + if target.is_dir(): + candidates = sorted(target.glob("*.tex")) + return candidates[0] if candidates else None + return None + + +def find_bib(tex: Path) -> Path | None: + bib = tex.with_name(tex.stem + ".bib") + if bib.exists(): + return bib + for candidate in sorted(tex.parent.glob("*.bib")): + return candidate + return None + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + prog="analise-artigo", + description="Analisa um artigo LaTeX (ABNT) e gera relatório de " + "estrutura, métricas e citações.", + ) + p.add_argument( + "artigo", + help="arquivo .tex ou diretório que contém o .tex", + ) + p.add_argument( + "--out", "-o", + default="relatorios", + help="diretório de saída do relatório (padrão: relatorios/)", + ) + args = p.parse_args(argv) + + target = Path(args.artigo) + tex = find_tex(target) + if tex is None: + print(f"erro: nenhum .tex encontrado em {target}", file=sys.stderr) + return 1 + + bib = find_bib(tex) + + print(f"→ artigo: {tex}") + print(f"→ referências: {bib or 'nenhum .bib encontrado'}") + + article = parse_article(tex) + refs = parse_bib(bib) if bib else {} + + analysis = analyze(article, refs) + + outdir = Path(args.out) + md, js = write_reports(analysis, outdir) + + resumo = analysis.metrics["resumo_checks"] + print(f"\nResumo: {resumo['ok']} ok, " + f"{resumo['warn']} aviso(s), " + f"{resumo['fail']} falha(s), " + f"{len(analysis.issues)} questão(ões) de prosa") + print(f"\n→ relatório: {md}") + print(f"→ dados: {js}") + return 0 if resumo["fail"] == 0 else 2 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/analise_artigo/analyze.py b/analise_artigo/analyze.py new file mode 100644 index 0000000..01d3d0f --- /dev/null +++ b/analise_artigo/analyze.py @@ -0,0 +1,346 @@ +"""Métricas textuais, consistência de citações e checklist de conformidade. + +Regras ABNT/NBR usadas: +- NBR 10520: citações no texto devem existir na referências e vice-versa; +- NBR 6022: elementos obrigatórios do artigo (título, resumo, + palavras-chave, seções, referências); +- NBR 14724: elementos pré/textuais/pós-textuais. +""" + +from __future__ import annotations + +import re +from collections import Counter +from dataclasses import dataclass, field + +from .bib import Reference +from .texparse import Article, Section, count_words + +# --------------------------------------------------------------------------- +# Estruturas de resultado +# --------------------------------------------------------------------------- + + +@dataclass +class Check: + code: str + label: str + status: str # ok | warn | fail + detail: str = "" + + @property + def icon(self) -> str: + return {"ok": "✅", "warn": "⚠️", "fail": "❌"}[self.status] + + +@dataclass +class Analysis: + article: Article + refs: dict[str, Reference] + checks: list[Check] = field(default_factory=list) + metrics: dict = field(default_factory=dict) + issues: list[str] = field(default_factory=list) + + +# --------------------------------------------------------------------------- +# Métricas de prosa +# --------------------------------------------------------------------------- + +VOWELS = r"[aeiouyáàâãéêëíîïóôõúü]" + + +def syllables(word: str) -> int: + """Aproximação de sílabas: 1 sílaba por grupo de vogais. + + Contar *grupos* (em vez de caracteres) evita inflar ditongos como + "ei", "ão", "au" ("eficiência" → e|i|ê|i|a = 5, correto). + """ + w = re.sub(r"[^a-zA-Zà-öø-ÿ]", "", word.lower()) + if not w: + return 1 + n = len(re.findall(VOWELS + "+", w)) + return max(1, n) + + +def sentences(text: str) -> list[str]: + """Divide em frases respeitando as quebras de linha do .tex. + + Nos artigos processados cada parágrafo/item de lista ocupa uma linha + própria (convenção verificada nas fontes), então linha vira fronteira + segura de frase; dentro da linha, pontuação [.!?] segue valendo. + """ + out: list[str] = [] + for line in text.splitlines(): + line = line.strip() + if not line: + continue + parts = re.split(r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ“\"(])", line) + for p in parts: + p = p.strip() + # ignora marcadores de lista ("-", "1.") e ruído trivial + if len(p) <= 1 or re.fullmatch(r"\d+[\.\)]|-+", p): + continue + out.append(p) + return out + + +def flesch(text: str) -> float | None: + """Índice de Flesch adaptado para o português (0-100; mais alto = mais legível). + + Fórmula padrão do português: 206,835 − 1,015·MPF − 84,6·VMP, com + sílabas estimadas por grupos de vogais (aproximação, sem acentuação + completa). Resultado clamped em [0, 100]. + """ + words = count_words(text) + sents = sentences(text) + if not words or not sents: + return None + syl = sum(syllables(w) for w in words) + mpf = len(words) / len(sents) # média de palavras por frase + vmp = syl / len(words) # vocabulário médio (sílabas/palavra) + score = 206.835 - 1.015 * mpf - 84.6 * vmp + return round(min(100.0, max(0.0, score)), 1) + + +def flesch_band(score: float | None) -> str: + if score is None: + return "n/d" + if score >= 80: + return "Muito fácil" + if score >= 60: + return "Fácil" + if score >= 40: + return "Médio" + if score >= 20: + return "Difícil" + return "Muito difícil" + + +# --------------------------------------------------------------------------- +# Heurísticas de qualidade +# --------------------------------------------------------------------------- + +REPEATED_WORD = re.compile(r"\b([a-zà-ÿ]{2,})\s+\1\b", re.I) +REDUNDANT = re.compile( + r"\b(a|ao|de|do|da|que|e|mas|no|na|em|por|para)\s+\1\b", re.I +) +LONG_SENTENCE = 60 # palavras +LONG_PARAGRAPH = 160 # palavras + + +def find_grammar_issues(text: str) -> list[str]: + issues: list[str] = [] + for m in REPEATED_WORD.finditer(text): + # "se se" é legítimo em português pronominal ("avaliou-se se cada + # estudo..."); filtrado para não gerar falso-positivo + if m.group(1).lower() == "se": + continue + issues.append( + f"PALAVRA REPLICADA: \"{m.group(0).strip()}\" — " + f"...{context_around(text, m.start())}..." + ) + for m in REDUNDANT.finditer(text): + issues.append( + f"REDUNDÂNCIA: \"{m.group(0).strip()}\" — " + f"...{context_around(text, m.start())}..." + ) + return issues + + +def context_around(text: str, pos: int, span: int = 60) -> str: + start = max(0, pos - span) + end = min(len(text), pos + span) + snippet = text[start:end].replace("\n", " ") + return ("…" + snippet) if start > 0 else snippet + + +def long_sentences(text: str, limit: int = LONG_SENTENCE) -> list[tuple[str, int]]: + out = [] + for s in sentences(text): + n = len(count_words(s)) + if n > limit: + out.append((s[:120] + ("…" if len(s) > 120 else ""), n)) + return sorted(out, key=lambda x: -x[1])[:10] + + +def long_paragraphs(text: str, limit: int = LONG_PARAGRAPH) -> list[tuple[str, int]]: + # Convenção das fontes analisadas: um parágrafo por linha no .tex, + # portanto cada linha (depois da limpeza) é candidata a parágrafo. + out = [] + for p in (line.strip() for line in text.splitlines()): + if not p: + continue + n = len(count_words(p)) + if n > limit: + out.append((p[:120] + "…", n)) + return sorted(out, key=lambda x: -x[1])[:10] + + +def frequent_phrases(text: str, n: int = 5, top: int = 12) -> list[tuple[str, int]]: + words = [w.lower() for w in count_words(text)] + stop = {"de", "da", "do", "que", "com", "uma", "uma", "para", "em", "os", + "as", "no", "na", "e", "ou", "ao", "à", "dos", "das", "é", "são", + "and", "or"} # operadores de busca do estilo PRISMA (AND/OR) + words = [w for w in words if w not in stop and len(w) > 2] + grams = [" ".join(words[i : i + n]) for i in range(len(words) - n + 1)] + return Counter(grams).most_common(top) + + +def citation_stats(article: Article, refs: dict[str, Reference]) -> dict: + cited = Counter(article.citations) + defined = set(refs) + used = set(cited) + return { + "total_citations": len(article.citations), + "unique_cited": len(used), + "defined_refs": len(defined), + "orphan_citations": sorted(used - defined), + "unused_refs": sorted(defined - used), + "most_cited": [ + (k, n, refs.get(k).label() if k in refs else "?") + for k, n in cited.most_common(8) + ], + "year_distribution": dict( + Counter( + (refs[k].year or "?") if k in refs else "?" for k in used + ) + ), + } + + +# --------------------------------------------------------------------------- +# Análise completa +# --------------------------------------------------------------------------- + +RECOGNIZED_SECTIONS = ["introdução", "metodologia", "conclusão"] +EXPECTED_LITERATURE = ["revisão de literatura", "revisão da literatura", + "estado da arte", "revisão de literatura", + "fundamentação", "referencial"] + + +def analyze(article: Article, refs: dict[str, Reference]) -> Analysis: + checks: list[Check] = [] + issues: list[str] = [] + + # --- estrutura / pré-textuais ---------------------------------------- + checks.append(Check( + "ABNT-01", "Título presente", + "ok" if article.title else "fail", + article.title or "não encontrado no \\titulo{}", + )) + checks.append(Check( + "ABNT-02", "Autores informados", + "ok" if article.authors else "fail", + ", ".join(article.authors) or "não encontrado", + )) + checks.append(Check( + "ABNT-03", "Local e data", + "ok" if article.local and article.data else "warn", + f"{article.local}, {article.data}".strip(",") or "incompleto", + )) + + abs_words = len(count_words(article.abstract)) + checks.append(Check( + "ABNT-04", "Resumo presente (150-500 palavras, NBR 6022)", + "ok" if 150 <= abs_words <= 500 else "warn", + f"{abs_words} palavras" if abs_words else "não encontrado", + )) + checks.append(Check( + "ABNT-05", "Palavras-chave (3 a 8, separadas por ponto)", + "ok" if 3 <= len(article.keywords) <= 8 else "warn", + ", ".join(article.keywords) or "não encontradas", + )) + + all_titles = [s.title.lower() for s in article.sections] + for expected in RECOGNIZED_SECTIONS: + hit = next((t for t in all_titles if expected in t), None) + checks.append(Check( + "ABNT-06", f"Seção '{expected.capitalize()}'", + "ok" if hit else "warn", + hit or "não identificada", + )) + lit_hit = next((t for t in all_titles + if any(e in t for e in EXPECTED_LITERATURE)), None) + checks.append(Check( + "ABNT-07", "Seção de revisão de literatura", + "ok" if lit_hit else "fail", + lit_hit or "faltando", + )) + + # citações / referências ------------------------------------------------ + stats = citation_stats(article, refs) + checks.append(Check( + "ABNT-08", "Citações no corpo do texto", + "ok" if stats["total_citations"] >= 5 else "warn", + f"{stats['total_citations']} citações, " + f"{stats['unique_cited']} referências citadas", + )) + checks.append(Check( + "ABNT-09", "Citações sem entrada no .bib (NBR 10520)", + "ok" if not stats["orphan_citations"] else "fail", + ", ".join(stats["orphan_citations"]) or "nenhuma", + )) + checks.append(Check( + "ABNT-10", "Entradas .bib não citadas no texto (NBR 10520)", + "ok" if not stats["unused_refs"] else "warn", + ", ".join(stats["unused_refs"]) or "nenhuma", + )) + + # figuras ------------------------------------------------------------ + checks.append(Check( + "ABNT-11", "Figuras/quadros identificados", + "ok" if (article.figures or article.tables) else "warn", + f"{len(article.figures)} figuras, {len(article.tables)} quadros", + )) + + # prosa --------------------------------------------------------------- + body_text = "\n\n".join(s.text for s in article.sections) + flesch_score = flesch(body_text) + top_phrases = frequent_phrases(body_text) + repeated = find_grammar_issues(article.abstract + " " + body_text) + l_sentences = long_sentences(body_text) + l_paragraphs = long_paragraphs(body_text) + for i in repeated: + issues.append(i) + for s, n in l_sentences: + issues.append(f"Frase longa ({n} palavras): \"{s}\"") + for p, n in l_paragraphs: + issues.append(f"Parágrafo extenso ({n} palavras): \"{p}\"") + + metrics = { + "palavras_corpo": len(count_words(body_text)), + "palavras_totais": article.text_words, + "sentencas": len(sentences(body_text)), + "media_palavras_por_sentenca": round( + len(count_words(body_text)) / max(1, len(sentences(body_text))), 1 + ), + "vocabulario_unico": len( + set(w.lower() for w in count_words(body_text)) + ), + "flesch": flesch_score, + "flesch_faixa": flesch_band(flesch_score), + "secoes": [ + { + "titulo": s.title, + "nivel": s.level, + "palavras": s.words, + "percentual": round( + 100 * s.words / max(1, article.text_words), 1 + ), + } + for s in article.sections + ], + "top_frases": top_phrases, + "citacoes": stats, + } + + summary = { + status: sum(1 for c in checks if c.status == status) + for status in ("ok", "warn", "fail") + } + metrics["resumo_checks"] = summary + + return Analysis( + article=article, refs=refs, checks=checks, metrics=metrics, + issues=issues, + ) diff --git a/analise_artigo/bib.py b/analise_artigo/bib.py new file mode 100644 index 0000000..f81a7d2 --- /dev/null +++ b/analise_artigo/bib.py @@ -0,0 +1,167 @@ +"""Parser simples de arquivos .bib (BibTeX). + +Usa apenas a biblioteca padrão; suporta campos com valores em chaves +aninhadas e valores multi-linha. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from pathlib import Path + + +@dataclass +class Reference: + key: str + type: str # article, inproceedings, misc, ... + fields: dict[str, str] = field(default_factory=dict) + + @property + def authors(self) -> str: + value = self.fields.get("author", "").strip() + # "X, A. and B, C. and others" -> "X, A., B, C. et al." (apenas campos author) + value = re.sub(r"\s+and\s+others\b", " et al.", value, flags=re.I) + return re.sub(r"\s+and\s+", ", ", value, flags=re.I) + + @property + def title(self) -> str: + return self.fields.get("title", "").strip() + + @property + def year(self) -> str: + return self.fields.get("year", "").strip() + + + @property + def journal(self) -> str: + return self.fields.get( + "journal", self.fields.get("booktitle", "") + ).strip() + + def label(self) -> str: + """Rótulo curto autor (ano).""" + first = re.split(r",|and", self.authors)[0].strip() + if "," in first: + surname = first.split(",")[0].strip() + else: + surname = last_token(first) + return f"{surname} ({self.year})" if self.year else surname + + +def last_token(name: str) -> str: + parts = name.replace("-", " ").split() + return parts[-1] if parts else name + + +def _split_entries(text: str) -> list[tuple[str, str, str]]: + """Retorna [(type, key, body), ...] respeitando aninhamento de chaves.""" + entries: list[tuple[str, str, str]] = [] + i = 0 + while True: + m = re.compile(r"@(\w+)\s*\{").search(text, i) + if not m: + break + etype, key = m.group(1), "" + start = m.end() + depth = 1 + j = start + while j < len(text) and depth: + if text[j] == "{": + depth += 1 + elif text[j] == "}": + depth -= 1 + j += 1 + body = text[start : j - 1] + + # chave = conteúdo até a primeira vírgula fora de chaves + depth = 0 + k = 0 + while k < len(body): + c = body[k] + if c == "{": + depth += 1 + elif c == "}": + depth -= 1 + elif c == "," and depth == 0: + break + if c == "\n" and depth == 0: + break + k += 1 + key = body[:k].strip() + entries.append((etype, key, body[k + 1 :])) + i = j + return entries + + +def _parse_fields(body: str) -> dict[str, str]: + fields: dict[str, str] = {} + i = 0 + n = len(body) + while i < n: + while i < n and body[i] in " \t\n,": + i += 1 + if i >= n: + break + m = re.compile(r"(\w+)\s*=").match(body, i) + if not m: + break + fname = m.group(1).lower() + i = m.end() + while i < n and body[i] in " \t\n": + i += 1 + if i >= n: + break + if body[i] == "{": + depth = 1 + j = i + 1 + while j < n and depth: + if body[j] == "{": + depth += 1 + elif body[j] == "}": + depth -= 1 + j += 1 + value = body[i + 1 : j - 1] + i = j + else: + m2 = re.compile(r"([^,]+)").match(body, i) + if not m2: + i += 1 + continue + value = m2.group(1) + i = m2.end() + value = value.strip() + value = re.sub(r"\s*\n\s*", " ", value) + value = value.replace("{", "").replace("}", "").strip() + value = _clean_bib_value(value) + fields[fname] = value + return fields + + +def _clean_bib_value(value: str) -> str: + """Remove escapes LaTeX correntes em .bib (diacríticos, \\& etc.). + + Ex.: ``Ki\\vs\\vs`` -> ``Kiss``, ``L'opez`` -> ``Lopez``, + ``Tirke\\cs`` -> ``Tirkes``. + """ + value = re.sub( + r"\\[\'\"~vc](?:\{([A-Za-z])\}|([A-Za-z]))", + lambda m: m.group(1) or m.group(2), + value, + flags=re.I, + ) + for esc, ch in (("\\&", "&"), ("\\$", "$"), ("\\%", "%"), + ("\\_", "_"), ("\\#", "#")): + value = value.replace(esc, ch) + value = value.replace("---", " — ").replace("--", " – ") + return value + + +def parse_bib(bib_path: Path | str) -> dict[str, Reference]: + text = Path(bib_path).read_text(encoding="utf-8") + refs: dict[str, Reference] = {} + for etype, key, body in _split_entries(text): + if not key: + continue + refs[key] = Reference(key=key, type=etype, fields=_parse_fields(body)) + return refs diff --git a/analise_artigo/report.py b/analise_artigo/report.py new file mode 100644 index 0000000..da9b93c --- /dev/null +++ b/analise_artigo/report.py @@ -0,0 +1,172 @@ +"""Gera o relatório em Markdown e o artefato JSON.""" + +from __future__ import annotations + +import json +import re +from datetime import datetime, timezone +from pathlib import Path + +from .analyze import Analysis +from .bib import Reference +from .texparse import count_words + + +def _md_escape(text: str) -> str: + return text.replace("|", "\\|") + + +def _ref_line(ref: Reference) -> str: + parts = [ref.year and ref.year, ref.title, ref.authors, ref.journal] + parts = [re.sub(r"\s*\n\s*", " ", p) for p in parts if p] + if not parts: + return ref.key + return "; ".join(parts) if len(parts) > 1 else parts[0] + + +def build_markdown(a: Analysis) -> str: + art, m = a.article, a.metrics + lines: list[str] = [] + add = lines.append + + add(f"# Análise do artigo") + add("") + add(f"> **{art.title}**") + add(f"> Autores: {', '.join(art.authors)} — {art.local}, {art.data} — {art.instituicao}") + add("") + + resumo = m["resumo_checks"] + add(f"**Resultado global:** ✅ {resumo['ok']} ok · ⚠️ {resumo['warn']} aviso(s) · ❌ {resumo['fail']} falha(s) · 📝 {len(a.issues)} questão(ões) de prosa") + add("") + + # 1. Estrutura --------------------------------------------------------- + add("## 1. Estrutura e elementos") + add("") + add("| # | Verificação | Status | Detalhe |") + add("|---|---|---|---|") + for c in a.checks: + add(f"| {c.code} | {c.label} | {c.icon} | {_md_escape(c.detail)} |") + add("") + + # 2. Métricas textuais ------------------------------------------------ + add("## 2. Métricas textuais") + add("") + add("| Métrica | Valor |") + add("|---|---|") + add(f"| Palavras (corpo) | {m['palavras_corpo']:,} |") + add(f"| Palavras (total, c/ resumo) | {m['palavras_totais']:,} |") + add(f"| Frases | {m['sentencas']} |") + add(f"| Média de palavras por frase | {m['media_palavras_por_sentenca']} |") + add(f"| Vocabulário único | {m['vocabulario_unico']:,} |") + add(f"| Legibilidade (Flesch adaptado) | {m['flesch']} — {m['flesch_faixa']} |") + add("") + + add("### Distribuição por seção") + add("") + add("| Seção | Palavras | % do total |") + add("|---|---:|---:|") + for s in m["secoes"]: + add(f"| {s['titulo']} | {s['palavras']:,} | {s['percentual']}% |") + add("") + + # 3. Citações ---------------------------------------------------------- + stats = m["citacoes"] + add("## 3. Citações e referências (NBR 10520 / 6023)") + add("") + add(f"- Citações no texto: **{stats['total_citations']}** ({stats['unique_cited']} referências distintas)") + add(f"- Entradas no `.bib`: **{stats['defined_refs']}**") + if stats["orphan_citations"]: + add(f"- ❌ Citadas mas **ausentes do .bib**: {', '.join(stats['orphan_citations'])}") + if stats["unused_refs"]: + add(f"- ⚠️ No .bib mas **não citadas no texto**: {', '.join(stats['unused_refs'])}") + add("") + if stats["most_cited"]: + add("### Referências mais citadas") + add("") + add("| Referência | Cit.ªs | Entrada |") + add("|---|---:|---|") + for key, n, label in stats["most_cited"]: + add(f"| {_md_escape(label)} | {n} | `{key}` |") + add("") + if stats["year_distribution"]: + add("### Referências por ano") + add("") + years = sorted(stats["year_distribution"].items()) + add("| Ano | N.º |") + add("|---:|---:|") + for year, n in years: + add(f"| {year} | {n} |") + add("") + + # 4. Prosa ------------------------------------------------------------- + add("## 4. Questões de prosa") + add("") + if a.issues: + for i in a.issues: + add(f"- {i}") + else: + add("Nenhuma questão detectada pelas heurísticas.") + add("") + if m["top_frases"]: + add("### Expressões mais repetidas (5 palavras)") + add("") + for phrase, n in m["top_frases"]: + add(f"- \"{phrase}\" — {n}×") + add("") + + # 5. Referências ------------------------------------------------------- + add("## 5. Referências do .bib") + add("") + for key in sorted(a.refs): + ref = a.refs[key] + cited = key in set(a.article.citations) + mark = "✅" if cited else "⚠️" + add(f"{mark} `{key}` — {_md_escape(_ref_line(ref))}") + add("") + + add("---") + add(f"_Gerado em {datetime.now(timezone.utc):%Y-%m-%d %H:%M} UTC " + f"pelo projeto `analise-artigo` (v1.0)._") + return "\n".join(lines) + "\n" + + +def build_json(a: Analysis) -> str: + d = { + "article": { + "title": a.article.title, + "authors": a.article.authors, + "local": a.article.local, + "data": a.article.data, + "instituicao": a.article.instituicao, + "keywords": a.article.keywords, + "abstract_words": len(count_words(a.article.abstract)), + "sections": [ + {"titulo": s.title, "nivel": s.level, "palavras": s.words} + for s in a.article.sections + ], + "figures": a.article.figures, + "tables": a.article.tables, + }, + "checks": [ + {"code": c.code, "label": c.label, "status": c.status, + "detail": c.detail} + for c in a.checks + ], + "metrics": a.metrics, + "issues": a.issues, + "references": { + k: {"type": r.type, "fields": r.fields, + "cited": k in set(a.article.citations)} + for k, r in a.refs.items() + }, + } + return json.dumps(d, ensure_ascii=False, indent=2) + + +def write_reports(a: Analysis, outdir: Path) -> tuple[Path, Path]: + outdir.mkdir(parents=True, exist_ok=True) + md_path = outdir / "relatorio.md" + json_path = outdir / "relatorio.json" + md_path.write_text(build_markdown(a), encoding="utf-8") + json_path.write_text(build_json(a), encoding="utf-8") + return md_path, json_path diff --git a/analise_artigo/texparse.py b/analise_artigo/texparse.py new file mode 100644 index 0000000..5292c89 --- /dev/null +++ b/analise_artigo/texparse.py @@ -0,0 +1,264 @@ +"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT). + +Converte o .tex em texto limpo preservando: +- título, autores, local/data (elementos pré-textuais); +- resumo e palavras-chave; +- seções/subseções e seus parágrafos; +- chaves de citação (\\cite, \\citeonline, \\citeauthor ...). + +Não depende de pandoc: usa apenas a biblioteca padrão. +""" + +from __future__ import annotations + +import re +from dataclasses import dataclass, field +from pathlib import Path + + +@dataclass +class Section: + """Seção do artigo com o texto limpo já convertido para prosa.""" + + title: str + level: int = 1 + + @property + def words(self) -> int: + return len(count_words(self.text)) + + +@dataclass +class Article: + title: str = "" + authors: list[str] = field(default_factory=list) + local: str = "" + data: str = "" + instituicao: str = "" + abstract: str = "" + keywords: list[str] = field(default_factory=list) + sections: list[Section] = field(default_factory=list) + citations: list[str] = field(default_factory=list) # ordem de aparição + figures: list[str] = field(default_factory=list) # captions das figuras + tables: list[str] = field(default_factory=list) # captions de quadros + text_words: int = 0 + + def section(self, title: str) -> Section | None: + for s in self.sections: + if s.title.strip().lower() == title.strip().lower(): + return s + return None + + +# --------------------------------------------------------------------------- +# Conversão LaTeX -> texto +# --------------------------------------------------------------------------- + +_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline", + "bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"] + +_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}") +_BRACE_CMD_RE = { + cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS +} +# Chaves com barra escapada no LaTeX (ex.: "\_", "\#"). +# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima. +_ESCAPE_MAP = { + "\\_": "_", + "\\#": "#", + "\\\\": "\n", +} + + +def count_words(text: str) -> list[str]: + return re.findall( + r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text + ) + + +def _fold_brace_commands(s: str) -> str: + changed = True + while changed: + changed = False + for cmd, pat in _BRACE_CMD_RE.items(): + s2, n = pat.subn(r"\1", s) + if n: + s, changed = s2, True + return s + + +def strip_latex(text: str, citations: list[str] | None = None) -> str: + """Remove comandos e ambientes, preservando o conteúdo legível.""" + + # citações: captura as chaves e remove do texto + def _cite(m: re.Match) -> str: + if citations is not None: + for key in m.group(1).split(","): + key = key.strip() + if key: + citations.append(key) + return " " + + text = _CITE_RE.sub(_cite, text) + + # rótulos/referências internas e figuras + text = re.sub(r"\\label\{[^}]*\}", " ", text) + text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text) + text = re.sub(r"\\includegraphics.*", " ", text) + text = re.sub(r"\\noindent", " ", text) + + # comandos com argumento em chaves genéricos (legend, caption, textbf...) + text = _fold_brace_commands(text) + text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text) + + # ambientes: manter marcadores de item, eliminar o resto + text = re.sub(r"\\begin\{itemize\}", "\n- ", text) + text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text) + text = re.sub(r"\\item\b", "\n- ", text) + text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text) + text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text) + text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor + + # comandos restantes sem argumento + text = re.sub(r"\\[a-zA-Z]+\*?", " ", text) + text = re.sub(r"\\[,;:\\!%$&]", " ", text) + for esc, ch in _ESCAPE_MAP.items(): + text = text.replace(esc, ch) + + # aspas e travessões + text = text.replace("``", '"').replace("''", '"') + text = text.replace("---", " — ").replace("--", " – ") + + # tabular: & vira coluna + text = text.replace("&", " | ") + + # limpar pontuação e espaços + text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb] + text = re.sub(r"[{}]+", " ", text) + text = re.sub(r"[ \t]+", " ", text) + text = re.sub(r" +([,.;:!?])", r"\1", text) + text = re.sub(r"\n\s*", "\n", text) + text = re.sub(r"\n{3,}", "\n\n", text) + return text.strip() + + +# --------------------------------------------------------------------------- +# Parsing da estrutura +# --------------------------------------------------------------------------- + +_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}") + + +def _capture_command(text: str, name: str) -> str: + """Extrai o último conteúdo de \\name{...} ou \\ ... \\end{}.""" + m = re.search(r"\\" + name + r"\s*\{", text) + if not m: + return "" + start = m.end() + depth = 1 + i = start + while i < len(text) and depth: + if text[i] == "{": + depth += 1 + elif text[i] == "}": + depth -= 1 + i += 1 + return text[start : i - 1].strip() + + +def _extract_authors(text: str) -> list[str]: + raw = _capture_command(text, "autor") + if not raw: + raw = _capture_command(text, "author") + return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()] + + +def parse_article(tex_path: Path | str) -> Article: + tex_path = Path(tex_path) + text = tex_path.read_text(encoding="utf-8") + + # remove comentários de linha (não escapados) + text = re.sub(r"(?=3.10). +# Nenhuma dependência necessária para o modo atual de uso. + +# (Opcional, para extensão futura) +# spacy>=3.7 # tokenização/POS pt-BR +# language-tool-off # correção gramatical via ponteiro