Web Scraping en Argentina: búsqueda de personas por DNI paso a paso
En este video te muestro paso a paso cómo hacer Web Scraping en Argentina usando Python, aplicándolo a la búsqueda de personas por DNI mediante fuentes públicas
link del video: https://www.youtube.com/watch?v=9SNqYQ2qAC4
**************************HTML*****************************
<!DOCTYPE html>
<html>
<head>
<meta name="viewport" content="width=device-width" />
<title>Index</title>
{% load static %}
<link rel="stylesheet" type="text/css" href="{% static 'styles/general.css' %}" />
<style>
.cntpreview{
background: #000;
padding: 10px;
width: 50%;
min-height: 300px;
color: #fff
}
</style>
</head>
<body>
<div id="divLoading" class="wrap_loading hide">
<div class="lds-ring">
<div></div>
<div></div>
<div></div>
<div></div>
</div>
<div class="loading_text">Procesando ...</div>
</div>
<div class="titulo">Consulta persona con DNI (Argentina)</div>
<div class="fila">
<input type="text" id="txtIdentificador" value="33016244,10433615,21834641"/>
<button id="btnProcesar" class="boton">Procesar</button>
</div>
<div style="display: flex;justify-content: center;">
<pre id="txtPre" class="cntpreview"></pre>
</div>
<div>
{% csrf_token %}
</div>
<script src="{% static 'scripts/Demo44.js' %}" ></script>
</body>
</html>
**************************JAVASCRIPT*****************************
window.onload=function(){
let btnProcesar=document.getElementById("btnProcesar");
btnProcesar.onclick=function(){
let txtIdentificador=document.getElementById("txtIdentificador").value;
let fd=new FormData();
fd.append("data",txtIdentificador)
servidor({url:"procesar",data:fd,responsetype:"json"}).then((data)=>{
document.getElementById("txtPre").innerHTML=JSON.stringify(data,null,2);
});
}
}
function servidor({ metodo = "post", url = null, data = null, responsetype = "text" } = {}) {
return new Promise((resolve, reject) => {
let divLoading=document.getElementById("divLoading");
if(divLoading){
divLoading.classList.remove("hide");
}
let xhr = new XMLHttpRequest();
xhr.open(metodo, url);
var csrftoken=document.getElementsByName("csrfmiddlewaretoken")[0].value;
xhr.setRequestHeader("X-CSRFToken", csrftoken);
xhr.responseType = responsetype;
xhr.onreadystatechange = function () {
if (xhr.readyState == 4 && xhr.status == 200) {
divLoading.classList.add("hide");
resolve(xhr.response);
}
}
xhr.onerror = function (e) {
reject(e)
}
xhr.send(data);
});
}
**************************PYTHON(DJANGO)*****************************
from django.shortcuts import render
from django.http.response import HttpResponse,JsonResponse
from django.conf import settings
import requests
import json
def index(request):
return render(request,"Demo44.html")
def procesar(request):
dni=request.POST.get("data")
rptaJson={}
print("buscar dni "+dni)
try:
header={"user-agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/143.0.0.0 Safari/537.36",
}
sesion=requests.session()
req=sesion.get("https://informes.nosis.com/",headers=header,verify=False)
if req.status_code==200:
header["origin"]="https://informes.nosis.com"
header["referer"]="https://informes.nosis.com/"
header["content-type"]="application/x-www-form-urlencoded; charset=UTF-8"
payload="Texto="+dni+"&Tipo=-1&EdadDesde=-1&EdadHasta=-1&IdProvincia=-1&Localidad=&recaptcha_response_field=enganio+al+captcha&recaptcha_challenge_field=enganio+al+captcha&encodedResponse="
req=sesion.post("https://informes.nosis.com/Home/Buscar",data=payload,headers=header,verify=False)
if req.status_code==200:
objrpta=req.json()
if "EntidadesEncontradas" in objrpta and len(objrpta["EntidadesEncontradas"])>0:
rptaJson=objrpta["EntidadesEncontradas"][0]
del rptaJson["UrlInforme"]
del rptaJson["UrlClon"]
except Exception as e:
print("Error "+ str(e))
return JsonResponse(rptaJson,safe=False)
Comentarios
Publicar un comentario