Web Scraping MASIVO CURP en México con Python
Aprende a hacer Web Scraping en Python desde cero utilizando únicamente la librería requests. En este tutorial te muestro cómo extraer datos como nombres, apellidos y fechas de nacimiento, tanto de forma individual como masiva.
****************************HTML***********************
<!DOCTYPE html>
<html>
<head>
<meta name="viewport" content="width=device-width" />
<title>Index</title>
{% load static %}
<link rel="stylesheet" type="text/css" href="{% static 'styles/general.css' %}" />
<style>
.cntpreview{
background: #000;
padding: 10px;
width: 50%;
min-height: 300px;
color: #fff
}
</style>
</head>
<body>
<div id="divLoading" class="wrap_loading hide">
<div class="lds-ring">
<div></div>
<div></div>
<div></div>
<div></div>
</div>
<div class="loading_text">Procesando ...</div>
</div>
<div class="titulo">Consulta personas con CURP Mexico</div>
<div class="fila">
<textarea style="width: 186px;height: 70px;" id="txtIdentificador" value="" ></textarea>
<button id="btnProcesar" class="boton">Procesar</button>
</div>
<div style="display: flex;justify-content: center;">
<pre id="txtPre" class="cntpreview"></pre>
</div>
<div>
{% csrf_token %}
</div>
<script src="{% static 'scripts/Demo68.js' %}" ></script>
</body>
</html>
****************************JAVASCRIPT***********************
window.onload=function(){
let btnProcesar=document.getElementById("btnProcesar");
btnProcesar.onclick=function(){
let txtFecha=document.getElementById("txtIdentificador").value;
let fd=new FormData();
fd.append("data",txtFecha);
servidor({url:"procesar",data:fd,responsetype:"json"}).then((data)=>{
document.getElementById("txtPre").innerHTML=JSON.stringify(data,null,2);
});
}
}
function servidor({ metodo = "post", url = null, data = null, responsetype = "text" } = {}) {
return new Promise((resolve, reject) => {
let divLoading=document.getElementById("divLoading");
if(divLoading){
divLoading.classList.remove("hide");
}
let xhr = new XMLHttpRequest();
xhr.open(metodo, url);
var csrftoken=document.getElementsByName("csrfmiddlewaretoken")[0].value;
xhr.setRequestHeader("X-CSRFToken", csrftoken);
xhr.responseType = responsetype;
xhr.onreadystatechange = function () {
if (xhr.readyState == 4 && xhr.status == 200) {
divLoading.classList.add("hide");
resolve(xhr.response);
}
}
xhr.onerror = function (e) {
reject(e)
}
xhr.send(data);
});
}
****************************PYTHON***********************
from django.shortcuts import render
from django.http.response import HttpResponse,JsonResponse
import requests
import json
requests.packages.urllib3.disable_warnings()
def index(request):
return render(request,"Demo68.html")
def procesar(request):
curps=request.POST.get("data")
rptaJson=[]
try:
acodigos=curps.split(",")
for curp in acodigos:
rptaJson.append( consultar(curp))
except Exception as e:
print("Error "+ str(e))
return JsonResponse(rptaJson,safe=False)
def consultar(curp):
objRpta={}
header={"user-agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/146.0.0.0 Safari/537.36",
"content-type":"application/json; charset=utf-8"}
sesion=requests.session()
payload={"curp":curp}
req=sesion.post("https://us-central1-os-gobierno-de-nuevo-leon.cloudfunctions.net/nuevoLeon-checkCurp",data=json.dumps(payload),headers=header,verify=False)
if req.status_code==200:
objRpta=req.json()
return objRpta
Comentarios
Publicar un comentario