All files / services/crawler-api/src/models CrawlJob.js

97.36% Statements 74/76
100% Branches 1/1
0% Functions 0/1
97.36% Lines 74/76

Press n or j to go to the next uncovered block, b, p or k for the previous block.

1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 771x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x     1x 1x 1x  
const mongoose = require('mongoose');
 
const crawlJobSchema = new mongoose.Schema({
  jobId: {
    type: String,
    required: true,
    unique: true,
    index: true,
  },
  configId: {
    type: mongoose.Schema.Types.ObjectId,
    ref: 'CrawlerConfig',
  },
  userId: {
    type: String,
    default: 'api-key',
    index: true,
  },
  organizationId: {
    type: String,
    required: true,
    index: true,
  },
  scheduleId: {
    type: String,
    default: null,
    index: true,
  },
  status: {
    type: String,
    enum: ['pending', 'running', 'completed', 'failed', 'cancelled'],
    default: 'pending',
  },
  config: {
    startUrls: [{ type: String }],
    maxDepth: Number,
    maxPages: Number,
    query: String,
    pageThreshold: Number,
    linkThreshold: Number,
    enableSynthesis: Boolean,
    synthetiser: mongoose.Schema.Types.Mixed,
    router: mongoose.Schema.Types.Mixed,
  },
  progress: {
    step: String,
    percent: Number,
    pagesCrawled: { type: Number, default: 0 },
    pagesRelevant: { type: Number, default: 0 },
    currentUrl: String,
  },
  results: {
    filePath: String,
    recordCount: Number,
    sizeBytes: Number,
  },
  errorMessage: String,
  logs: [{
    level: String,
    message: String,
    context: mongoose.Schema.Types.Mixed,
    timestamp: { type: Date, default: Date.now },
  }],
}, {
  timestamps: true,
});
 
crawlJobSchema.index({ organizationId: 1, createdAt: -1 });
crawlJobSchema.index({ userId: 1, createdAt: -1 });
 
crawlJobSchema.methods.addLog = async function(level, message, context = {}) {
  this.logs.push({ level, message, context, timestamp: new Date() });
  await this.save();
};
 
module.exports = mongoose.model('CrawlJob', crawlJobSchema, 'crawljobs');