Press n or j to go to the next uncovered block, b, p or k for the previous block.
| 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 | 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x 1x | const mongoose = require('mongoose');
const crawlJobSchema = new mongoose.Schema({
jobId: {
type: String,
required: true,
unique: true,
index: true,
},
configId: {
type: mongoose.Schema.Types.ObjectId,
ref: 'CrawlerConfig',
},
userId: {
type: String,
default: 'api-key',
index: true,
},
organizationId: {
type: String,
required: true,
index: true,
},
scheduleId: {
type: String,
default: null,
index: true,
},
status: {
type: String,
enum: ['pending', 'running', 'completed', 'failed', 'cancelled'],
default: 'pending',
},
config: {
startUrls: [{ type: String }],
maxDepth: Number,
maxPages: Number,
query: String,
pageThreshold: Number,
linkThreshold: Number,
enableSynthesis: Boolean,
synthetiser: mongoose.Schema.Types.Mixed,
router: mongoose.Schema.Types.Mixed,
},
progress: {
step: String,
percent: Number,
pagesCrawled: { type: Number, default: 0 },
pagesRelevant: { type: Number, default: 0 },
currentUrl: String,
},
results: {
filePath: String,
recordCount: Number,
sizeBytes: Number,
},
errorMessage: String,
logs: [{
level: String,
message: String,
context: mongoose.Schema.Types.Mixed,
timestamp: { type: Date, default: Date.now },
}],
}, {
timestamps: true,
});
crawlJobSchema.index({ organizationId: 1, createdAt: -1 });
crawlJobSchema.index({ userId: 1, createdAt: -1 });
crawlJobSchema.methods.addLog = async function(level, message, context = {}) {
this.logs.push({ level, message, context, timestamp: new Date() });
await this.save();
};
module.exports = mongoose.model('CrawlJob', crawlJobSchema, 'crawljobs');
|